Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0d3d3a3855 |
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "ltx-core"
|
||||
version = "1.1.6"
|
||||
version = "v1.1.6"
|
||||
description = "Core implementation of Lightricks' LTX-2 model"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.10"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "ltx-pipelines"
|
||||
version = "1.1.6"
|
||||
version = "v1.1.6"
|
||||
description = "Pipelines implementation for Lightricks' LTX-2 model"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.10"
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
configs/*.yaml
|
||||
datasets
|
||||
outputs
|
||||
wandb
|
||||
|
||||
@@ -1,340 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Audio-to-Audio IC-LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training In-Context LoRA (IC-LoRA) adapters that
|
||||
# enable audio-to-audio transformations. IC-LoRA learns to apply audio
|
||||
# transformations (e.g., style transfer, voice conversion, sound effects, etc.)
|
||||
# by conditioning on reference audio.
|
||||
#
|
||||
# Key differences from text-to-video LoRA:
|
||||
# - Uses reference audio as conditioning input alongside text prompts
|
||||
# - Requires preprocessed reference audio latents in addition to target latents
|
||||
# - Audio-only training (no video modality)
|
||||
# - Validation requires reference audio to demonstrate the transformation
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── conditions/ # Text embeddings for each sample
|
||||
# ├── audio_latents/ # Target audio latents (what the model learns to generate)
|
||||
# └── reference_audio_latents/ # Reference audio latents (conditioning input)
|
||||
#
|
||||
# Dataset metadata columns: audio, reference_audio, caption
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 16-32 for IC-LoRA.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES (not used for audio-only IC-LoRA):
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (not used for audio-only IC-LoRA):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# For audio-only IC-LoRA, we explicitly target audio modules.
|
||||
# Including audio FFN layers often improves transformation quality.
|
||||
target_modules:
|
||||
# Audio self-attention
|
||||
- "audio_attn1.to_k"
|
||||
- "audio_attn1.to_q"
|
||||
- "audio_attn1.to_v"
|
||||
- "audio_attn1.to_out.0"
|
||||
# Audio cross-attention to text
|
||||
- "audio_attn2.to_k"
|
||||
- "audio_attn2.to_q"
|
||||
- "audio_attn2.to_v"
|
||||
- "audio_attn2.to_out.0"
|
||||
# Audio feed-forward (often improves transformation quality)
|
||||
- "audio_ff.net.0.proj"
|
||||
- "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the audio-to-audio (IC-LoRA) training approach using the unified
|
||||
# flexible strategy. Reference conditioning concatenates pre-encoded reference
|
||||
# audio latents to the target sequence. Reference tokens participate in
|
||||
# bidirectional self-attention but receive no noise and are excluded from loss.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Audio modality configuration (audio-only, no video)
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing target audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# Conditions applied to the audio modality during training
|
||||
conditions:
|
||||
# Reference conditioning (IC-LoRA): concatenates pre-encoded reference audio
|
||||
# latents to the target sequence. The model learns to transform the reference
|
||||
# into the target audio based on the text prompt.
|
||||
- type: reference
|
||||
# Directory name (within preprocessed_data_root) containing reference audio latents
|
||||
# These are the conditioning inputs that guide the transformation
|
||||
latents_dir: "reference_audio_latents"
|
||||
# Probability of applying reference conditioning per training sample
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: audio_latents/, conditions/, and reference_audio_latents/ subdirectories
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# For IC-LoRA, each sample includes a reference condition pointing to the conditioning audio.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A warm and soothing piano melody with soft ambient textures, reminiscent of a quiet
|
||||
evening by the fireplace. Gentle reverb creates a sense of intimate space.
|
||||
conditions:
|
||||
- type: reference
|
||||
audio: "/path/to/reference_audio_1.wav"
|
||||
- prompt: >-
|
||||
Energetic electronic beats with pulsing synthesizer leads and crisp hi-hat patterns.
|
||||
Deep bass tones provide a driving rhythm underneath bright melodic arpeggios.
|
||||
conditions:
|
||||
- type: reference
|
||||
audio: "/path/to/reference_audio_2.wav"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Generation length control [width, height, frames]
|
||||
# With generate_video=false, width/height are unused; frames and frame_rate set audio duration.
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" skips both audio and video self-attention (suited to audio validation)
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Can be enabled even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Whether to generate video in validation samples
|
||||
# Disabled for audio-only IC-LoRA since no video modality is configured
|
||||
generate_video: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "ic-lora", "a2a" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/a2a_ic_lora"
|
||||
@@ -1,344 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Audio-to-Video LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# audio-to-video generation. The model learns to generate videos conditioned
|
||||
# on a frozen audio signal via the transformer's built-in cross-modal attention.
|
||||
#
|
||||
# In this mode, audio is provided as a frozen (clean, no noise, no loss)
|
||||
# conditioning signal. The video modality is the only generated output.
|
||||
# Audio influences video generation through the transformer's audio-to-video
|
||||
# cross-attention mechanism.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Generate videos driven by audio content (e.g., music visualizations)
|
||||
# - Train audio-reactive video generation
|
||||
# - Create models that synchronize video with given audio
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (frozen conditioning input)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# Using short patterns like "to_k" matches ALL attention modules (video, audio, and cross-modal).
|
||||
# For audio-video training, this is the recommended approach.
|
||||
target_modules:
|
||||
# Attention layers (matches both video and audio branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
# - "audio_ff.net.0.proj"
|
||||
# - "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the audio-to-video training approach using the unified flexible strategy.
|
||||
# Audio is frozen (no noise, sigma=0, excluded from loss) and conditions video
|
||||
# generation via the transformer's built-in cross-modal attention.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
# Video is the generated (denoised) output
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Audio modality configuration
|
||||
# Audio is frozen — it acts as conditioning for video generation
|
||||
# Frozen modalities get sigma=0, timestep=0, no noise, and no loss
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
# When false, audio is passed through the transformer clean and influences
|
||||
# video via cross-modal attention
|
||||
is_generated: false
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, audio_latents/, and conditions/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A musician plays an acoustic guitar in a dimly lit recording studio, fingers moving
|
||||
across the fretboard with practiced ease. The warm amber light from a desk lamp
|
||||
illuminates the wooden guitar body. Sound-absorbing panels line the walls, and a
|
||||
microphone stands nearby on a boom arm.
|
||||
conditions:
|
||||
- type: audio_to_video
|
||||
audio: "/path/to/conditioning_audio_1.wav"
|
||||
- prompt: >-
|
||||
Rain falls steadily on a cobblestone street in a European town at dusk. Puddles form
|
||||
between the stones, creating ripples as new drops land. Old brick buildings line both
|
||||
sides of the narrow street, their facades glistening with moisture under warm
|
||||
streetlights.
|
||||
conditions:
|
||||
- type: audio_to_video
|
||||
audio: "/path/to/conditioning_audio_2.wav"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_v" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy.audio.is_generated - you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
# For this audio-to-video config, false matches frozen audio conditioning (video-only synthesis).
|
||||
generate_audio: false
|
||||
|
||||
# Generate video from the frozen audio condition
|
||||
generate_video: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "a2v" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/a2v_lora"
|
||||
@@ -1,343 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Audio Extension (Forward) LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# forward audio extension. The model learns to continue audio forward in time
|
||||
# by conditioning on a prefix of existing audio latent timesteps.
|
||||
#
|
||||
# Prefix conditioning works by providing the first N audio latent timesteps as
|
||||
# clean conditioning signals during training — they receive no noise,
|
||||
# timestep=0, and are excluded from the loss. The model learns to generate
|
||||
# audio that seamlessly continues from the given prefix.
|
||||
#
|
||||
# This is an audio-only training mode — no video modality is configured.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to extend/continue existing audio
|
||||
# - Fine-tune audio temporal continuation capabilities
|
||||
# - Create seamless audio extension models
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── conditions/ # Text embeddings for each sample
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for audio extension LoRA training.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES (not used for audio-only modes):
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (not used for audio-only modes):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# For audio-only extension, we explicitly target audio modules.
|
||||
# Including audio FFN layers can increase the LoRA's capacity.
|
||||
target_modules:
|
||||
# Audio self-attention
|
||||
- "audio_attn1.to_k"
|
||||
- "audio_attn1.to_q"
|
||||
- "audio_attn1.to_v"
|
||||
- "audio_attn1.to_out.0"
|
||||
# Audio cross-attention to text
|
||||
- "audio_attn2.to_k"
|
||||
- "audio_attn2.to_q"
|
||||
- "audio_attn2.to_v"
|
||||
- "audio_attn2.to_out.0"
|
||||
# Audio feed-forward (often improves transformation quality)
|
||||
- "audio_ff.net.0.proj"
|
||||
- "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the audio extension training approach using the unified flexible
|
||||
# strategy. Prefix conditioning provides the first N audio latent timesteps as
|
||||
# clean conditioning, teaching the model to generate temporal continuations.
|
||||
training_strategy:
|
||||
name: "flexible"
|
||||
|
||||
# Audio modality configuration (audio-only, no video)
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# Conditions applied to the audio modality during training
|
||||
conditions:
|
||||
- type: prefix
|
||||
# Number of audio latent timesteps to use as conditioning prefix
|
||||
# Each audio latent timestep = 1 patchified token
|
||||
temporal_boundary: 8
|
||||
# Probability of applying prefix conditioning per training sample
|
||||
# At 1.0, all training samples use audio extension mode
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: conditions/ and audio_latents/ subdirectories
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Each sample includes an audio prefix condition for forward audio extension.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A warm and soothing piano melody with soft ambient textures, reminiscent of a quiet
|
||||
evening by the fireplace. Gentle reverb creates a sense of intimate space.
|
||||
conditions:
|
||||
- type: prefix
|
||||
audio: "/path/to/prefix_audio_1.wav"
|
||||
# Duration is in seconds; choose a value that covers a comparable audio context
|
||||
# to the training temporal_boundary measured in audio latent timesteps.
|
||||
duration: 1.0
|
||||
- prompt: >-
|
||||
Energetic electronic beats with pulsing synthesizer leads and crisp hi-hat patterns.
|
||||
Deep bass tones provide a driving rhythm underneath bright melodic arpeggios.
|
||||
conditions:
|
||||
- type: prefix
|
||||
audio: "/path/to/prefix_audio_2.wav"
|
||||
# Duration is in seconds; choose a value that covers a comparable audio context
|
||||
# to the training temporal_boundary measured in audio latent timesteps.
|
||||
duration: 1.0
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Generation length control [width, height, frames]
|
||||
# With generate_video=false, width/height are unused; frames and frame_rate set audio duration.
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" skips both audio and video self-attention (suited to audio validation)
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Enabled because this audio-only config generates audio
|
||||
generate_audio: true
|
||||
|
||||
# Whether to generate video in validation samples
|
||||
# Disabled because no video modality is configured
|
||||
generate_video: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "audio-extension", "audio-only" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/audio_extend_lora"
|
||||
@@ -1,324 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Audio Inpainting LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# audio inpainting. The model learns to fill in masked regions of audio
|
||||
# using per-sample binary masks that define which regions are conditioning
|
||||
# (provided clean) and which regions the model must generate.
|
||||
#
|
||||
# Mask conditioning works by loading per-sample binary masks from disk.
|
||||
# Masked regions receive clean latents (no noise, timestep=0) and are excluded
|
||||
# from the loss. Unmasked regions are noised and trained normally.
|
||||
#
|
||||
# This is an audio-only training mode — no video modality is configured.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to fill in or replace regions of existing audio
|
||||
# - Fine-tune audio inpainting capabilities on custom datasets
|
||||
# - Create audio restoration or editing models
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── conditions/ # Text embeddings for each sample
|
||||
# ├── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
# └── audio_masks/ # Per-sample binary masks defining conditioning regions
|
||||
#
|
||||
# Dataset metadata columns: audio, audio_mask, caption
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for audio inpainting LoRA training.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# For audio-only inpainting, we explicitly target audio modules.
|
||||
# Including audio FFN layers can increase the LoRA's capacity.
|
||||
target_modules:
|
||||
# Audio self-attention
|
||||
- "audio_attn1.to_k"
|
||||
- "audio_attn1.to_q"
|
||||
- "audio_attn1.to_v"
|
||||
- "audio_attn1.to_out.0"
|
||||
# Audio cross-attention to text
|
||||
- "audio_attn2.to_k"
|
||||
- "audio_attn2.to_q"
|
||||
- "audio_attn2.to_v"
|
||||
- "audio_attn2.to_out.0"
|
||||
# Audio feed-forward (often improves transformation quality)
|
||||
- "audio_ff.net.0.proj"
|
||||
- "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the audio inpainting training approach using the unified flexible
|
||||
# strategy. Per-sample binary masks define which audio regions are provided
|
||||
# as clean conditioning and which regions the model must learn to generate.
|
||||
training_strategy:
|
||||
name: "flexible"
|
||||
|
||||
# Audio modality configuration (audio-only, no video)
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# Conditions applied to the audio modality during training
|
||||
conditions:
|
||||
- type: mask
|
||||
# Directory name (within preprocessed_data_root) containing binary masks
|
||||
# Each mask file corresponds to a training sample and defines the
|
||||
# conditioning region (mask=1 means conditioning, mask=0 means generate)
|
||||
mask_dir: "audio_masks"
|
||||
# Probability of applying mask conditioning per training sample
|
||||
# At 1.0, all training samples use inpainting mode
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: conditions/, audio_latents/, and audio_masks/ subdirectories
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# For audio inpainting, each sample includes mask conditioning (audio + mask paths).
|
||||
samples:
|
||||
- prompt: >-
|
||||
A warm and soothing piano melody with soft ambient textures, reminiscent of a quiet
|
||||
evening by the fireplace. Gentle reverb creates a sense of intimate space.
|
||||
conditions:
|
||||
- type: mask
|
||||
audio: "/path/to/inpainting_audio_1.wav"
|
||||
mask: "/path/to/inpainting_mask_1.pt"
|
||||
- prompt: >-
|
||||
Energetic electronic beats with pulsing synthesizer leads and crisp hi-hat patterns.
|
||||
Deep bass tones provide a driving rhythm underneath bright melodic arpeggios.
|
||||
conditions:
|
||||
- type: mask
|
||||
audio: "/path/to/inpainting_audio_2.wav"
|
||||
mask: "/path/to/inpainting_mask_2.pt"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Generation length control [width, height, frames]
|
||||
# With generate_video=false, width/height are unused; frames and frame_rate set audio duration.
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" skips both audio and video self-attention (suited to audio validation)
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Enabled because this audio-only config generates audio
|
||||
generate_audio: true
|
||||
|
||||
# Whether to generate video in validation samples
|
||||
# Disabled because no video modality is configured
|
||||
generate_video: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "audio-inpainting", "audio-only" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/audio_inpainting_lora"
|
||||
@@ -1,343 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Audio Extension (Backward) LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# backward audio extension. The model learns to generate audio content leading into existing audio
|
||||
# by conditioning on a suffix of existing audio latent timesteps.
|
||||
#
|
||||
# Suffix conditioning works by providing the last N audio latent timesteps as
|
||||
# clean conditioning signals during training — they receive no noise,
|
||||
# timestep=0, and are excluded from the loss. The model learns to generate
|
||||
# audio that seamlessly leads into the given suffix.
|
||||
#
|
||||
# This is an audio-only training mode — no video modality is configured.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to generate preceding content for existing audio
|
||||
# - Fine-tune audio temporal continuation capabilities
|
||||
# - Create seamless audio extension models
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── conditions/ # Text embeddings for each sample
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for audio extension LoRA training.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES (not used for audio-only modes):
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (not used for audio-only modes):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# For audio-only extension, we explicitly target audio modules.
|
||||
# Including audio FFN layers can increase the LoRA's capacity.
|
||||
target_modules:
|
||||
# Audio self-attention
|
||||
- "audio_attn1.to_k"
|
||||
- "audio_attn1.to_q"
|
||||
- "audio_attn1.to_v"
|
||||
- "audio_attn1.to_out.0"
|
||||
# Audio cross-attention to text
|
||||
- "audio_attn2.to_k"
|
||||
- "audio_attn2.to_q"
|
||||
- "audio_attn2.to_v"
|
||||
- "audio_attn2.to_out.0"
|
||||
# Audio feed-forward (often improves transformation quality)
|
||||
- "audio_ff.net.0.proj"
|
||||
- "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the backward audio extension training approach using the unified flexible
|
||||
# strategy. Suffix conditioning provides the last N audio latent timesteps as
|
||||
# clean conditioning, teaching the model to generate content leading into existing audio.
|
||||
training_strategy:
|
||||
name: "flexible"
|
||||
|
||||
# Audio modality configuration (audio-only, no video)
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# Conditions applied to the audio modality during training
|
||||
conditions:
|
||||
- type: suffix
|
||||
# Number of audio latent timesteps to use as conditioning suffix
|
||||
# Each audio latent timestep = 1 patchified token
|
||||
temporal_boundary: 8
|
||||
# Probability of applying suffix conditioning per training sample
|
||||
# At 1.0, all training samples use audio extension mode
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: conditions/ and audio_latents/ subdirectories
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Each sample includes an audio suffix condition for backward audio extension.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A warm and soothing piano melody with soft ambient textures, reminiscent of a quiet
|
||||
evening by the fireplace. Gentle reverb creates a sense of intimate space.
|
||||
conditions:
|
||||
- type: suffix
|
||||
audio: "/path/to/suffix_audio_1.wav"
|
||||
# Duration is in seconds; choose a value that covers a comparable audio context
|
||||
# to the training temporal_boundary measured in audio latent timesteps.
|
||||
duration: 1.0
|
||||
- prompt: >-
|
||||
Energetic electronic beats with pulsing synthesizer leads and crisp hi-hat patterns.
|
||||
Deep bass tones provide a driving rhythm underneath bright melodic arpeggios.
|
||||
conditions:
|
||||
- type: suffix
|
||||
audio: "/path/to/suffix_audio_2.wav"
|
||||
# Duration is in seconds; choose a value that covers a comparable audio context
|
||||
# to the training temporal_boundary measured in audio latent timesteps.
|
||||
duration: 1.0
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Generation length control [width, height, frames]
|
||||
# With generate_video=false, width/height are unused; frames and frame_rate set audio duration.
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" skips both audio and video self-attention (suited to audio validation)
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Enabled because this audio-only config generates audio
|
||||
generate_audio: true
|
||||
|
||||
# Whether to generate video in validation samples
|
||||
# Disabled because no video modality is configured
|
||||
generate_video: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "audio-suffix", "audio-only" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/audio_suffix_lora"
|
||||
@@ -1,342 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 AV2AV IC-LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training In-Context LoRA (IC-LoRA) adapters that
|
||||
# enable joint audio-video-to-audio-video transformations. IC-LoRA learns to
|
||||
# apply transformations to both the video and audio modalities simultaneously
|
||||
# by conditioning on paired reference video and audio.
|
||||
#
|
||||
# Both modalities use reference conditioning: pre-encoded reference latents
|
||||
# are concatenated to each modality's target sequence. Reference tokens
|
||||
# participate in bidirectional self-attention but receive no noise and are
|
||||
# excluded from the loss.
|
||||
#
|
||||
# Key differences from video-only IC-LoRA (v2v_ic_lora.yaml):
|
||||
# - Both video AND audio have reference conditions
|
||||
# - Requires preprocessed reference latents for BOTH modalities
|
||||
# - LoRA targets all modules (video, audio, and cross-modal attention)
|
||||
# - Validation uses both video and audio reference conditions
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Target video latents
|
||||
# ├── audio_latents/ # Target audio latents
|
||||
# ├── conditions/ # Text embeddings
|
||||
# ├── reference_latents/ # Reference video latents (conditioning input)
|
||||
# └── reference_audio_latents/ # Reference audio latents (conditioning input)
|
||||
#
|
||||
# Dataset metadata columns: video, audio, reference_video, reference_audio, caption
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# IC-LoRA reference conditioning is intended for LoRA adapter training.
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 16-32 for IC-LoRA.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# For AV2AV IC-LoRA, we target ALL modules — video, audio, and cross-modal attention.
|
||||
# Using short patterns matches all branches simultaneously.
|
||||
target_modules:
|
||||
# Attention layers (matches video, audio, and cross-modal branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the AV2AV IC-LoRA training approach using the unified flexible
|
||||
# strategy. Both video and audio modalities have reference conditioning,
|
||||
# enabling joint audiovisual transformations.
|
||||
training_strategy:
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing target video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
conditions:
|
||||
# Reference conditioning (IC-LoRA): concatenates pre-encoded reference video
|
||||
# latents to the target sequence
|
||||
- type: reference
|
||||
latents_dir: "reference_latents"
|
||||
probability: 1.0
|
||||
|
||||
# Audio modality configuration
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing target audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# Conditions applied to the audio modality during training
|
||||
conditions:
|
||||
# Reference conditioning (IC-LoRA): concatenates pre-encoded reference audio
|
||||
# latents to the target sequence
|
||||
- type: reference
|
||||
latents_dir: "reference_audio_latents"
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, audio_latents/, conditions/, reference_latents/, and reference_audio_latents/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# For AV2AV IC-LoRA, each sample includes reference video and reference audio conditions.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A man in a casual blue jacket walks along a winding path through a lush green park on a
|
||||
bright sunny afternoon. Tall oak trees line the pathway, their leaves rustling gently in
|
||||
the breeze. Dappled sunlight creates shifting patterns on the ground as he strolls at a
|
||||
relaxed pace, occasionally looking up at the scenery around him. The audio captures
|
||||
footsteps on gravel, birds singing in the trees, distant children playing, and the soft
|
||||
whisper of wind through the foliage.
|
||||
conditions:
|
||||
- type: reference
|
||||
video: "/path/to/reference_video_1.mp4"
|
||||
downscale_factor: 1
|
||||
temporal_scale_factor: 1
|
||||
include_in_output: true
|
||||
- type: reference
|
||||
audio: "/path/to/reference_audio_1.wav"
|
||||
- prompt: >-
|
||||
A fluffy orange tabby cat sits perfectly still on a wooden windowsill, its green eyes
|
||||
intently tracking small birds hopping on a branch just outside the glass. The cat's ears
|
||||
twitch and rotate, following every movement. Warm afternoon light illuminates its fur,
|
||||
creating a soft golden glow. Behind the cat, a cozy living room with a bookshelf and
|
||||
houseplants is visible. The audio features gentle purring, occasional soft meows, muffled
|
||||
bird chirps through the window, and quiet ambient room sounds.
|
||||
conditions:
|
||||
- type: reference
|
||||
video: "/path/to/reference_video_2.mp4"
|
||||
downscale_factor: 1
|
||||
temporal_scale_factor: 1
|
||||
include_in_output: true
|
||||
- type: reference
|
||||
audio: "/path/to/reference_audio_2.wav"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # Both video and audio modalities are trained
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Can be enabled even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "ic-lora", "av2av" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/av2av_ic_lora"
|
||||
@@ -1,348 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Image-to-Video LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# image-to-video generation. The model learns to generate videos conditioned
|
||||
# on a starting image (first frame), with optional audio generation.
|
||||
#
|
||||
# First-frame conditioning works by providing the first frame as a clean
|
||||
# conditioning signal during training — it receives no noise, timestep=0,
|
||||
# and is excluded from the loss. This teaches the model to animate from
|
||||
# a given image.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to generate videos starting from a given image
|
||||
# - Fine-tune image-to-video capabilities on custom datasets
|
||||
# - Create image animation models with optional audio
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# Using short patterns like "to_k" matches ALL attention modules (video, audio, and cross-modal).
|
||||
# For audio-video training, this is the recommended approach.
|
||||
target_modules:
|
||||
# Attention layers (matches both video and audio branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
# - "audio_ff.net.0.proj"
|
||||
# - "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the image-to-video training approach using the unified flexible strategy.
|
||||
# First-frame conditioning provides the first frame as clean conditioning signal
|
||||
# during training, teaching the model to animate from a given image.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
# First-frame conditioning: the first frame of each video is provided as a clean
|
||||
# conditioning signal (no noise, timestep=0, excluded from loss)
|
||||
conditions:
|
||||
- type: first_frame
|
||||
# Probability of applying first-frame conditioning per training sample
|
||||
# At 0.5, half the training samples use I2V mode, half use pure T2V
|
||||
# Higher values improve I2V quality but may reduce T2V diversity
|
||||
probability: 0.5
|
||||
|
||||
# Audio modality configuration
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and audio_latents/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration. The audio captures the
|
||||
gentle clicking of keyboard keys, the soft rustle of papers, and ambient room tone with
|
||||
occasional distant bird chirps from outside.
|
||||
conditions:
|
||||
- type: first_frame
|
||||
image_or_video: "/path/to/conditioning_image_1.png"
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him. The audio features the sizzling of pans,
|
||||
the clinking of utensils against plates, and the ambient hum of kitchen ventilation.
|
||||
conditions:
|
||||
- type: first_frame
|
||||
image_or_video: "/path/to/conditioning_image_2.png"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy.audio.is_generated - you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "i2v" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/i2v_lora"
|
||||
@@ -1,317 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Text-to-Audio LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# text-to-audio generation. The model learns to generate audio from text
|
||||
# prompts without any additional conditioning.
|
||||
#
|
||||
# This is the simplest audio-only training mode — no reference audio, no
|
||||
# video modality. Only the audio branch of the transformer is trained.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Fine-tune audio generation for specific sound styles or domains
|
||||
# - Train custom audio generation capabilities
|
||||
# - Create audio LoRAs that can be combined with video LoRAs
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── conditions/ # Text embeddings for each sample
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general audio LoRA training.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES (not used for audio-only modes):
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (not used for audio-only modes):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# For audio-only training, we explicitly target audio modules.
|
||||
# Including audio FFN layers can increase the LoRA's capacity.
|
||||
target_modules:
|
||||
# Audio self-attention
|
||||
- "audio_attn1.to_k"
|
||||
- "audio_attn1.to_q"
|
||||
- "audio_attn1.to_v"
|
||||
- "audio_attn1.to_out.0"
|
||||
# Audio cross-attention to text
|
||||
- "audio_attn2.to_k"
|
||||
- "audio_attn2.to_q"
|
||||
- "audio_attn2.to_v"
|
||||
- "audio_attn2.to_out.0"
|
||||
# Audio feed-forward (often improves transformation quality)
|
||||
- "audio_ff.net.0.proj"
|
||||
- "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the text-to-audio training approach using the unified flexible
|
||||
# strategy. Audio-only training with no additional conditions — the model
|
||||
# learns to generate audio purely from text prompts.
|
||||
training_strategy:
|
||||
name: "flexible"
|
||||
|
||||
# Audio modality configuration (audio-only, no video)
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: conditions/ and audio_latents/ subdirectories
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Text-to-audio validation samples do not need additional conditions.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A warm and soothing piano melody with soft ambient textures, reminiscent of a quiet
|
||||
evening by the fireplace. Gentle reverb creates a sense of intimate space.
|
||||
- prompt: >-
|
||||
Energetic electronic beats with pulsing synthesizer leads and crisp hi-hat patterns.
|
||||
Deep bass tones provide a driving rhythm underneath bright melodic arpeggios.
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Generation length control [width, height, frames]
|
||||
# With generate_video=false, width/height are unused; frames and frame_rate set audio duration.
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" skips both audio and video self-attention (suited to audio validation)
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Enabled because this audio-only config generates audio
|
||||
generate_audio: true
|
||||
|
||||
# Whether to generate video in validation samples
|
||||
# Disabled because no video modality is configured
|
||||
generate_video: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "t2a", "audio-only" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/t2a_lora"
|
||||
@@ -1,327 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Text-to-Video LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# text-to-video generation with joint audio-video support.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Fine-tune LTX-2 on your own video dataset
|
||||
# - Train joint audio-video generation from text prompts
|
||||
# - Create custom video generation styles or audiovisual concepts
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# Using short patterns like "to_k" matches ALL attention modules (video, audio, and cross-modal).
|
||||
# For audio-video training, this is the recommended approach.
|
||||
target_modules:
|
||||
# Attention layers (matches both video and audio branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
# - "audio_ff.net.0.proj"
|
||||
# - "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the training approach using the unified flexible strategy.
|
||||
# This configuration trains both video and audio generation from text prompts.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
# When is_generated is true, the model learns to generate (denoise) video
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Audio modality configuration
|
||||
# When is_generated is true, the model learns to generate (denoise) audio
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and audio_latents/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration. The audio captures the
|
||||
gentle clicking of keyboard keys, the soft rustle of papers, and ambient room tone with
|
||||
occasional distant bird chirps from outside.
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him. The audio features the sizzling of pans,
|
||||
the clinking of utensils against plates, and the ambient hum of kitchen ventilation.
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy.audio.is_generated - you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "t2v" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/t2v_lora"
|
||||
@@ -1,339 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Text-to-Video LoRA Training Configuration (Low VRAM)
|
||||
# =============================================================================
|
||||
#
|
||||
# This is a memory-optimized variant of the standard text-to-video LoRA config.
|
||||
# It uses 8-bit optimizer, int8 quantization, and reduced LoRA rank to minimize
|
||||
# GPU memory usage while maintaining good training quality.
|
||||
#
|
||||
# Memory optimizations applied:
|
||||
# - 8-bit AdamW optimizer (reduces optimizer state memory by ~75%)
|
||||
# - INT8 model quantization (reduces model memory by ~50%)
|
||||
# - Lower LoRA rank (16 vs 32, reduces trainable parameters)
|
||||
# - Gradient checkpointing enabled
|
||||
#
|
||||
# Recommended for GPUs with 32GB VRAM (e.g., RTX 5090).
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Fine-tune LTX-2 on your own video dataset with limited GPU memory
|
||||
# - Train joint audio-video generation from text prompts
|
||||
# - Create custom video generation styles or audiovisual concepts
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
# Using a lower rank (16) to reduce trainable parameters and memory usage.
|
||||
# This still provides good capacity for many fine-tuning tasks.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Using 16 for low VRAM configuration.
|
||||
rank: 16
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 16
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# Using short patterns like "to_k" matches ALL attention modules (video, audio, and cross-modal).
|
||||
# For audio-video training, this is the recommended approach.
|
||||
target_modules:
|
||||
# Attention layers (matches both video and audio branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
# - "audio_ff.net.0.proj"
|
||||
# - "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the training approach using the unified flexible strategy.
|
||||
# This configuration trains both video and audio generation from text prompts.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
# When is_generated is true, the model learns to generate (denoise) video
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Audio modality configuration
|
||||
# When is_generated is true, the model learns to generate (denoise) audio
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
# Using 8-bit AdamW to reduce optimizer state memory by ~75%
|
||||
optimizer_type: "adamw8bit"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
# Using INT8 quantization to reduce base model memory consumption by ~50%
|
||||
quantization: "int8-quanto"
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: true
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and audio_latents/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration. The audio captures the
|
||||
gentle clicking of keyboard keys, the soft rustle of papers, and ambient room tone with
|
||||
occasional distant bird chirps from outside.
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him. The audio features the sizzling of pans,
|
||||
the clinking of utensils against plates, and the ambient hum of kitchen ventilation.
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 49 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [ 29 ] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy audio modality settings — you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "t2v", "low-vram" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/t2v_lora_low_vram"
|
||||
@@ -1,349 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Video-to-Audio (Foley) LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# video-to-audio (Foley) generation. The model learns to generate audio
|
||||
# conditioned on a frozen video signal via the transformer's built-in
|
||||
# cross-modal attention.
|
||||
#
|
||||
# In this mode, video is provided as a frozen (clean, no noise, no loss)
|
||||
# conditioning signal. The audio modality is the only generated output.
|
||||
# Video influences audio generation through the transformer's video-to-audio
|
||||
# cross-attention mechanism.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Generate sound effects (Foley) for existing videos
|
||||
# - Train audio generation conditioned on visual content
|
||||
# - Create models that produce audio matching video content
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (frozen conditioning input)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio, generated output)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# For video-to-audio training, we target audio modules and the cross-modal
|
||||
# attention that allows audio to attend to video features.
|
||||
target_modules:
|
||||
# Audio self-attention
|
||||
- "audio_attn1.to_k"
|
||||
- "audio_attn1.to_q"
|
||||
- "audio_attn1.to_v"
|
||||
- "audio_attn1.to_out.0"
|
||||
# Audio cross-attention to text
|
||||
- "audio_attn2.to_k"
|
||||
- "audio_attn2.to_q"
|
||||
- "audio_attn2.to_v"
|
||||
- "audio_attn2.to_out.0"
|
||||
# Audio feed-forward
|
||||
- "audio_ff.net.0.proj"
|
||||
- "audio_ff.net.2"
|
||||
# Cross-modal attention: allows audio to attend to video features
|
||||
- "video_to_audio_attn.to_k"
|
||||
- "video_to_audio_attn.to_q"
|
||||
- "video_to_audio_attn.to_v"
|
||||
- "video_to_audio_attn.to_out.0"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the video-to-audio (Foley) training approach using the unified
|
||||
# flexible strategy. Video is frozen (no noise, sigma=0, excluded from loss)
|
||||
# and conditions audio generation via the transformer's cross-modal attention.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
# Video is frozen — it acts as conditioning for audio generation
|
||||
# Frozen modalities get sigma=0, timestep=0, no noise, and no loss
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
# When false, video is passed through the transformer clean and influences
|
||||
# audio via cross-modal attention
|
||||
is_generated: false
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Audio modality configuration
|
||||
# Audio is the generated (denoised) output
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, audio_latents/, and conditions/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
The sound of ocean waves crashing against rocky cliffs, with seagulls calling in the
|
||||
distance and wind whistling through coastal grass.
|
||||
conditions:
|
||||
- type: video_to_audio
|
||||
video: "/path/to/conditioning_video_1.mp4"
|
||||
- prompt: >-
|
||||
Footsteps echo in a marble hallway as a person walks steadily, with the distant hum of
|
||||
air conditioning and occasional door closing sounds.
|
||||
conditions:
|
||||
- type: video_to_audio
|
||||
video: "/path/to/conditioning_video_2.mp4"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Validation dimensions [width, height, frames]
|
||||
# Width/height resize the frozen video condition; frames and frame_rate set audio duration.
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy.audio.is_generated - you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Whether to generate video in validation samples
|
||||
# Disabled for V2A since video is frozen conditioning, not generated
|
||||
generate_video: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "v2a", "foley" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/v2a_lora"
|
||||
@@ -1,359 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Video-to-Video (IC-LoRA) Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training In-Context LoRA (IC-LoRA) adapters that
|
||||
# enable video-to-video transformations. IC-LoRA learns to apply visual
|
||||
# transformations (e.g., depth-to-video, pose control, style transfer, etc.)
|
||||
# by conditioning on reference videos.
|
||||
#
|
||||
# Key differences from text-to-video LoRA:
|
||||
# - Uses reference videos as conditioning input alongside text prompts
|
||||
# - Requires preprocessed reference latents in addition to target latents
|
||||
# - Validation requires reference videos to demonstrate the transformation
|
||||
#
|
||||
# Dataset structure for IC-LoRA training:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Target video latents (what the model learns to generate)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── reference_latents/ # Reference video latents (conditioning input)
|
||||
#
|
||||
# Dataset metadata columns: video, reference_video, caption
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# IC-LoRA reference conditioning is intended for LoRA adapter training.
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 16-32 for IC-LoRA.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES (not used for video-only IC-LoRA):
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction, not used for video-only IC-LoRA):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# For IC-LoRA (video-only), we explicitly target video modules.
|
||||
# Including FFN layers often improves transformation quality.
|
||||
target_modules:
|
||||
# Video self-attention
|
||||
- "attn1.to_k"
|
||||
- "attn1.to_q"
|
||||
- "attn1.to_v"
|
||||
- "attn1.to_out.0"
|
||||
# Video cross-attention
|
||||
- "attn2.to_k"
|
||||
- "attn2.to_q"
|
||||
- "attn2.to_v"
|
||||
- "attn2.to_out.0"
|
||||
# Video feed-forward (often improves transformation quality)
|
||||
- "ff.net.0.proj"
|
||||
- "ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the video-to-video (IC-LoRA) training approach using the unified
|
||||
# flexible strategy. Reference conditioning concatenates pre-encoded reference
|
||||
# video latents to the target sequence. Reference tokens participate in
|
||||
# bidirectional self-attention but receive no noise and are excluded from loss.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing target video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
conditions:
|
||||
# Reference conditioning (IC-LoRA): concatenates pre-encoded reference video
|
||||
# latents to the target sequence. The model learns to transform the reference
|
||||
# into the target video based on the text prompt.
|
||||
- type: reference
|
||||
# Directory name (within preprocessed_data_root) containing reference video latents
|
||||
# These are the conditioning inputs that guide the transformation
|
||||
latents_dir: "reference_latents"
|
||||
# Probability of applying reference conditioning per training sample
|
||||
probability: 1.0
|
||||
|
||||
# Optional first-frame conditioning to improve I2V capabilities
|
||||
# At low probability, this teaches the model to also accept first-frame input
|
||||
- type: first_frame
|
||||
probability: 0.2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 2e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 3000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and reference_latents/ subdirectories
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# For IC-LoRA, each sample includes a reference condition pointing to the conditioning video.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A man in a casual blue jacket walks along a winding path through a lush green park on a
|
||||
bright sunny afternoon. Tall oak trees line the pathway, their leaves rustling gently in
|
||||
the breeze. Dappled sunlight creates shifting patterns on the ground as he strolls at a
|
||||
relaxed pace, occasionally looking up at the scenery around him. The audio captures
|
||||
footsteps on gravel, birds singing in the trees, distant children playing, and the soft
|
||||
whisper of wind through the foliage.
|
||||
conditions:
|
||||
- type: reference
|
||||
video: "/path/to/reference_video_1.mp4"
|
||||
# Set these to match --reference-downscale-factor / --reference-temporal-scale-factor
|
||||
# if reference latents were preprocessed at reduced spatial or temporal resolution.
|
||||
downscale_factor: 1
|
||||
temporal_scale_factor: 1
|
||||
include_in_output: true
|
||||
- prompt: >-
|
||||
A fluffy orange tabby cat sits perfectly still on a wooden windowsill, its green eyes
|
||||
intently tracking small birds hopping on a branch just outside the glass. The cat's ears
|
||||
twitch and rotate, following every movement. Warm afternoon light illuminates its fur,
|
||||
creating a soft golden glow. Behind the cat, a cozy living room with a bookshelf and
|
||||
houseplants is visible. The audio features gentle purring, occasional soft meows, muffled
|
||||
bird chirps through the window, and quiet ambient room sounds.
|
||||
conditions:
|
||||
- type: reference
|
||||
video: "/path/to/reference_video_2.mp4"
|
||||
# Set these to match --reference-downscale-factor / --reference-temporal-scale-factor
|
||||
# if reference latents were preprocessed at reduced spatial or temporal resolution.
|
||||
downscale_factor: 1
|
||||
temporal_scale_factor: 1
|
||||
include_in_output: true
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 512, 512, 81 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_v" # "stg_v" for video-only (no audio training)
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Can be enabled even when not training the audio branch
|
||||
generate_audio: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: 3
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-ic-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "ic-lora", "v2v" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/v2v_ic_lora"
|
||||
@@ -1,355 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Video Extension LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# video extension (temporal continuation). The model learns to extend a video
|
||||
# forward in time by conditioning on a prefix of existing frames.
|
||||
#
|
||||
# Prefix conditioning works by providing the first N temporal units as clean
|
||||
# conditioning signals during training — they receive no noise, timestep=0,
|
||||
# and are excluded from the loss. The model learns to generate continuation
|
||||
# frames that seamlessly follow the given prefix.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to extend/continue existing videos
|
||||
# - Fine-tune temporal continuation capabilities on custom datasets
|
||||
# - Create seamless video extension models with optional audio
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# Using short patterns like "to_k" matches ALL attention modules (video, audio, and cross-modal).
|
||||
# For audio-video training, this is the recommended approach.
|
||||
target_modules:
|
||||
# Attention layers (matches both video and audio branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
# - "audio_ff.net.0.proj"
|
||||
# - "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the video extension training approach using the unified flexible strategy.
|
||||
# Prefix conditioning provides the first N latent frames as clean conditioning,
|
||||
# teaching the model to generate temporally coherent continuations.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
# Prefix conditioning: the first N temporal units are provided as clean
|
||||
# conditioning signals (no noise, timestep=0, excluded from loss)
|
||||
conditions:
|
||||
- type: prefix
|
||||
# Number of latent frames to use as conditioning prefix
|
||||
# For prefix conditioning, N latent frames correspond to (N - 1) * 8 + 1 pixel frames.
|
||||
# temporal_boundary=8 means 57 pixel frames are used as prefix.
|
||||
temporal_boundary: 8
|
||||
# Probability of applying prefix conditioning per training sample
|
||||
# At 1.0, all training samples use video extension mode
|
||||
probability: 1.0
|
||||
|
||||
# Audio modality configuration
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and audio_latents/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration. The audio captures the
|
||||
gentle clicking of keyboard keys, the soft rustle of papers, and ambient room tone with
|
||||
occasional distant bird chirps from outside.
|
||||
conditions:
|
||||
- type: prefix
|
||||
video: "/path/to/prefix_video_1.mp4"
|
||||
# Matches temporal_boundary=8 during training: (8 - 1) * 8 + 1 = 57 frames
|
||||
num_frames: 57
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him. The audio features the sizzling of pans,
|
||||
the clinking of utensils against plates, and the ambient hum of kitchen ventilation.
|
||||
conditions:
|
||||
- type: prefix
|
||||
video: "/path/to/prefix_video_2.mp4"
|
||||
# Matches temporal_boundary=8 during training: (8 - 1) * 8 + 1 = 57 frames
|
||||
num_frames: 57
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy.audio.is_generated - you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "video-extension" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/video_extend_lora"
|
||||
@@ -1,343 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Video Inpainting LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# video inpainting. The model learns to fill in masked regions of a video
|
||||
# using per-sample binary masks that define which regions are conditioning
|
||||
# (provided clean) and which regions the model must generate.
|
||||
#
|
||||
# Mask conditioning works by loading per-sample binary masks from disk.
|
||||
# Masked regions receive clean latents (no noise, timestep=0) and are excluded
|
||||
# from the loss. Unmasked regions are noised and trained normally.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to fill in or replace regions of existing videos
|
||||
# - Fine-tune video inpainting capabilities on custom datasets
|
||||
# - Create object removal or region editing models
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── video_masks/ # Per-sample binary masks defining conditioning regions
|
||||
#
|
||||
# Dataset metadata columns: video, video_mask, caption
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# This is a video-only config, so explicitly target video attention modules.
|
||||
target_modules:
|
||||
- "attn1.to_k"
|
||||
- "attn1.to_q"
|
||||
- "attn1.to_v"
|
||||
- "attn1.to_out.0"
|
||||
- "attn2.to_k"
|
||||
- "attn2.to_q"
|
||||
- "attn2.to_v"
|
||||
- "attn2.to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the video inpainting training approach using the unified flexible
|
||||
# strategy. Per-sample binary masks define which regions are provided as clean
|
||||
# conditioning and which regions the model must learn to generate.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
# Mask conditioning: per-sample binary masks loaded from disk define which
|
||||
# tokens are conditioning (clean, timestep=0, no loss) vs generated
|
||||
conditions:
|
||||
- type: mask
|
||||
# Directory name (within preprocessed_data_root) containing binary masks
|
||||
# Each mask file corresponds to a training sample and defines the
|
||||
# conditioning region (mask=1 means conditioning, mask=0 means generate)
|
||||
mask_dir: "video_masks"
|
||||
# Probability of applying mask conditioning per training sample
|
||||
# At 1.0, all training samples use inpainting mode
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and video_masks/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration.
|
||||
conditions:
|
||||
- type: mask
|
||||
video: "/path/to/inpainting_video_1.mp4"
|
||||
mask: "/path/to/inpainting_mask_1.mp4"
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him.
|
||||
conditions:
|
||||
- type: mask
|
||||
video: "/path/to/inpainting_video_2.mp4"
|
||||
mask: "/path/to/inpainting_mask_2.mp4"
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_v" # "stg_v" for video-only validation
|
||||
|
||||
# Video inpainting trains video only; do not generate validation audio.
|
||||
generate_audio: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "inpainting" ]
|
||||
|
||||
# Log validation outputs to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/video_inpainting_lora"
|
||||
@@ -1,341 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Video Outpainting LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# video outpainting (spatial extension). The model learns to generate content
|
||||
# surrounding a known rectangular region of the video.
|
||||
#
|
||||
# Spatial crop conditioning works by providing a rectangular pixel region as
|
||||
# clean conditioning during training — those tokens receive no noise,
|
||||
# timestep=0, and are excluded from the loss. The model learns to generate
|
||||
# the content outside the given region, seamlessly extending the scene.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Extend video content beyond its original boundaries
|
||||
# - Train spatial outpainting capabilities on custom datasets
|
||||
# - Create video aspect ratio conversion models
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# └── conditions/ # Text embeddings for each video
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# This is a video-only config, so explicitly target video attention modules.
|
||||
target_modules:
|
||||
- "attn1.to_k"
|
||||
- "attn1.to_q"
|
||||
- "attn1.to_v"
|
||||
- "attn1.to_out.0"
|
||||
- "attn2.to_k"
|
||||
- "attn2.to_q"
|
||||
- "attn2.to_v"
|
||||
- "attn2.to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the video outpainting training approach using the unified flexible
|
||||
# strategy. Spatial crop conditioning provides a rectangular pixel region as
|
||||
# clean conditioning, teaching the model to generate surrounding content.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
# Spatial crop conditioning provides a rectangular region as clean context;
|
||||
# the model learns to generate the surrounding video tokens.
|
||||
conditions:
|
||||
- type: spatial_crop
|
||||
# Rectangular pixel region provided as clean conditioning (y1, x1, y2, x2)
|
||||
# Pixels within this region are conditioning (clean, timestep=0, no loss)
|
||||
# Pixels outside this region are generated by the model
|
||||
# Coordinates are in pixel space — automatically converted to latent space
|
||||
spatial_region: [0, 0, 288, 576]
|
||||
# Probability of applying spatial crop conditioning per training sample
|
||||
# At 1.0, all training samples use outpainting mode
|
||||
probability: 1.0
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/ and conditions/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration.
|
||||
conditions:
|
||||
- type: spatial_crop
|
||||
video: "/path/to/outpainting_video_1.mp4"
|
||||
spatial_region: [0, 0, 288, 576]
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him.
|
||||
conditions:
|
||||
- type: spatial_crop
|
||||
video: "/path/to/outpainting_video_2.mp4"
|
||||
spatial_region: [0, 0, 288, 576]
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_v" # "stg_v" for video-only validation
|
||||
|
||||
# Video outpainting trains video only; do not generate validation audio.
|
||||
generate_audio: false
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "outpainting", "spatial-crop" ]
|
||||
|
||||
# Log validation outputs to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/video_outpainting_lora"
|
||||
@@ -1,355 +0,0 @@
|
||||
# =============================================================================
|
||||
# LTX-2 Video Backward Extension (Suffix) LoRA Training Configuration
|
||||
# =============================================================================
|
||||
#
|
||||
# This configuration is for training LoRA adapters on the LTX-2 model for
|
||||
# backward video extension. The model learns to generate video content that
|
||||
# leads into a given suffix of existing frames.
|
||||
#
|
||||
# Suffix conditioning works by providing the last N temporal units as clean
|
||||
# conditioning signals during training — they receive no noise, timestep=0,
|
||||
# and are excluded from the loss. The model learns to generate preceding
|
||||
# frames that seamlessly lead into the given suffix.
|
||||
#
|
||||
# Use this configuration when you want to:
|
||||
# - Train the model to generate "prequel" content for existing videos
|
||||
# - Fine-tune backward temporal continuation capabilities
|
||||
# - Create video backward extension models with optional audio
|
||||
#
|
||||
# Dataset structure:
|
||||
# preprocessed_data_root/
|
||||
# ├── latents/ # Video latents (VAE-encoded videos)
|
||||
# ├── conditions/ # Text embeddings for each video
|
||||
# └── audio_latents/ # Audio latents (VAE-encoded audio)
|
||||
#
|
||||
# =============================================================================
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Model Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the base model to fine-tune and the training mode.
|
||||
model:
|
||||
# Path to the LTX-2 model checkpoint (.safetensors file)
|
||||
# This should be a local path to your downloaded model
|
||||
model_path: "path/to/ltx-2-model.safetensors"
|
||||
|
||||
# Path to the text encoder model directory
|
||||
# For LTX-2, this is typically the Gemma-based text encoder
|
||||
text_encoder_path: "path/to/gemma-text-encoder"
|
||||
|
||||
# Training mode: "lora" for efficient adapter training, "full" for full fine-tuning
|
||||
# LoRA is recommended for most use cases (faster, less memory, prevents overfitting)
|
||||
training_mode: "lora"
|
||||
|
||||
# Optional: Path to resume training from a checkpoint
|
||||
# Can be a checkpoint file (.safetensors) or directory (uses latest checkpoint)
|
||||
load_checkpoint: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# LoRA Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the Low-Rank Adaptation parameters for efficient fine-tuning.
|
||||
lora:
|
||||
# Rank of the LoRA matrices (higher = more capacity but more parameters)
|
||||
# Typical values: 8, 16, 32, 64. Start with 32 for general fine-tuning.
|
||||
rank: 32
|
||||
|
||||
# Alpha scaling factor (usually set equal to rank)
|
||||
# The effective scaling is alpha/rank, so alpha=rank means scaling of 1.0
|
||||
alpha: 32
|
||||
|
||||
# Dropout probability for LoRA layers (0.0 = no dropout)
|
||||
# Can help with regularization if overfitting occurs
|
||||
dropout: 0.0
|
||||
|
||||
# Which transformer modules to apply LoRA to
|
||||
# The LTX-2 transformer has separate attention and FFN blocks for video and audio:
|
||||
#
|
||||
# VIDEO MODULES:
|
||||
# - attn1.to_k, attn1.to_q, attn1.to_v, attn1.to_out.0 (video self-attention)
|
||||
# - attn2.to_k, attn2.to_q, attn2.to_v, attn2.to_out.0 (video cross-attention to text)
|
||||
# - ff.net.0.proj, ff.net.2 (video feed-forward)
|
||||
#
|
||||
# AUDIO MODULES:
|
||||
# - audio_attn1.to_k, audio_attn1.to_q, audio_attn1.to_v, audio_attn1.to_out.0 (audio self-attention)
|
||||
# - audio_attn2.to_k, audio_attn2.to_q, audio_attn2.to_v, audio_attn2.to_out.0 (audio cross-attention to text)
|
||||
# - audio_ff.net.0.proj, audio_ff.net.2 (audio feed-forward)
|
||||
#
|
||||
# AUDIO-VIDEO CROSS-ATTENTION MODULES (for cross-modal interaction):
|
||||
# - audio_to_video_attn.to_k, audio_to_video_attn.to_q, audio_to_video_attn.to_v, audio_to_video_attn.to_out.0
|
||||
# (Q from video, K/V from audio - allows video to attend to audio features)
|
||||
# - video_to_audio_attn.to_k, video_to_audio_attn.to_q, video_to_audio_attn.to_v, video_to_audio_attn.to_out.0
|
||||
# (Q from audio, K/V from video - allows audio to attend to video features)
|
||||
#
|
||||
# Using short patterns like "to_k" matches ALL attention modules (video, audio, and cross-modal).
|
||||
# For audio-video training, this is the recommended approach.
|
||||
target_modules:
|
||||
# Attention layers (matches both video and audio branches)
|
||||
- "to_k"
|
||||
- "to_q"
|
||||
- "to_v"
|
||||
- "to_out.0"
|
||||
# Uncomment below to also train feed-forward layers (can increase the LoRA's capacity):
|
||||
# - "ff.net.0.proj"
|
||||
# - "ff.net.2"
|
||||
# - "audio_ff.net.0.proj"
|
||||
# - "audio_ff.net.2"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Training Strategy Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Defines the backward video extension training approach using the unified
|
||||
# flexible strategy. Suffix conditioning provides the last N latent frames as
|
||||
# clean conditioning, teaching the model to generate content leading into them.
|
||||
training_strategy:
|
||||
# Strategy name: "flexible" for the unified conditioning framework
|
||||
# Supports all training modes (T2V, I2V, V2V, A2V, V2A, etc.) through
|
||||
# modality-specific configuration blocks.
|
||||
name: "flexible"
|
||||
|
||||
# Video modality configuration
|
||||
video:
|
||||
# Whether the model generates video (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing video latents
|
||||
latents_dir: "latents"
|
||||
|
||||
# Conditions applied to the video modality during training
|
||||
# Suffix conditioning: the last N temporal units are provided as clean
|
||||
# conditioning signals (no noise, timestep=0, excluded from loss)
|
||||
conditions:
|
||||
- type: suffix
|
||||
# Number of latent frames to use as conditioning suffix
|
||||
# For suffix conditioning, N latent frames correspond to N * 8 pixel frames.
|
||||
# temporal_boundary=8 means the last 64 pixel frames are used as suffix.
|
||||
temporal_boundary: 8
|
||||
# Probability of applying suffix conditioning per training sample
|
||||
# At 1.0, all training samples use backward extension mode
|
||||
probability: 1.0
|
||||
|
||||
# Audio modality configuration
|
||||
audio:
|
||||
# Whether the model generates audio (true) or uses it as frozen conditioning (false)
|
||||
is_generated: true
|
||||
# Directory name (within preprocessed_data_root) containing audio latents
|
||||
latents_dir: "audio_latents"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Optimization Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls the training optimization parameters.
|
||||
optimization:
|
||||
# Learning rate for the optimizer
|
||||
# Typical range for LoRA: 1e-5 to 1e-4
|
||||
learning_rate: 1e-4
|
||||
|
||||
# Total number of training steps
|
||||
steps: 2000
|
||||
|
||||
# Batch size per GPU
|
||||
# Reduce if running out of memory
|
||||
batch_size: 1
|
||||
|
||||
# Number of gradient accumulation steps
|
||||
# Effective batch size = batch_size * gradient_accumulation_steps * num_gpus
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
# Maximum gradient norm for clipping (helps training stability)
|
||||
max_grad_norm: 1.0
|
||||
|
||||
# Optimizer type: "adamw" (standard) or "adamw8bit" (memory-efficient)
|
||||
optimizer_type: "adamw"
|
||||
|
||||
# Learning rate scheduler type
|
||||
# Options: "constant", "linear", "cosine", "cosine_with_restarts", "polynomial"
|
||||
scheduler_type: "linear"
|
||||
|
||||
# Additional scheduler parameters (depends on scheduler_type)
|
||||
scheduler_params: { }
|
||||
|
||||
# Enable gradient checkpointing to reduce memory usage
|
||||
# Recommended for training with limited GPU memory
|
||||
enable_gradient_checkpointing: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Acceleration Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hardware acceleration and memory optimization settings.
|
||||
acceleration:
|
||||
# Mixed precision training mode
|
||||
# Options: "no" (fp32), "fp16" (half precision), "bf16" (bfloat16, recommended)
|
||||
mixed_precision_mode: "bf16"
|
||||
|
||||
# Model quantization for reduced memory usage
|
||||
# Options: null (none), "int8-quanto", "int4-quanto", "int2-quanto", "fp8-quanto", "fp8uz-quanto"
|
||||
quantization: null
|
||||
|
||||
# Load text encoder in 8-bit precision to save memory
|
||||
# Useful when GPU memory is limited
|
||||
load_text_encoder_in_8bit: false
|
||||
|
||||
# Offload optimizer state to CPU during validation video sampling and restore it after.
|
||||
# Frees VRAM for the VAE decoder when optimizer state is large (full fine-tune, high-rank
|
||||
# LoRA). No effect under FSDP (sharded state).
|
||||
offload_optimizer_during_validation: false
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Data Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Specifies the training data location and loading parameters.
|
||||
data:
|
||||
# Root directory containing preprocessed training data
|
||||
# Should contain: latents/, conditions/, and audio_latents/
|
||||
preprocessed_data_root: "/path/to/preprocessed/data"
|
||||
|
||||
# Number of worker processes for data loading
|
||||
# Used for parallel data loading to speed up data loading
|
||||
num_dataloader_workers: 2
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Validation Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls validation sampling during training.
|
||||
# NOTE: Validation sampling use simplified inference pipelines and prioritizes speed over
|
||||
# maximum quality. For production-quality inference, use `packages/ltx-pipelines`.
|
||||
validation:
|
||||
# Validation samples — each sample describes a self-contained generation request.
|
||||
# Use 'conditions' to add conditioning (first_frame, prefix, suffix, reference, video_to_audio, etc.)
|
||||
# See docs/configuration-reference.md#validation-condition-types for the full list of condition types.
|
||||
samples:
|
||||
- prompt: >-
|
||||
A woman with long brown hair sits at a wooden desk in a cozy home office, typing on a
|
||||
laptop while occasionally glancing at notes beside her. Soft natural light streams through
|
||||
a large window, casting warm shadows across the room. She pauses to take a sip from a
|
||||
ceramic mug, then continues working with focused concentration. The audio captures the
|
||||
gentle clicking of keyboard keys, the soft rustle of papers, and ambient room tone with
|
||||
occasional distant bird chirps from outside.
|
||||
conditions:
|
||||
- type: suffix
|
||||
video: "/path/to/suffix_video_1.mp4"
|
||||
# Matches temporal_boundary=8 during training: 8 * 8 = 64 frames
|
||||
num_frames: 64
|
||||
- prompt: >-
|
||||
A chef in a white uniform stands in a professional kitchen, carefully plating a gourmet
|
||||
dish with precise movements. Steam rises from freshly cooked vegetables as he arranges
|
||||
them with tweezers. The stainless steel surfaces gleam under bright overhead lights, and
|
||||
various pots simmer on the stove behind him. The audio features the sizzling of pans,
|
||||
the clinking of utensils against plates, and the ambient hum of kitchen ventilation.
|
||||
conditions:
|
||||
- type: suffix
|
||||
video: "/path/to/suffix_video_2.mp4"
|
||||
# Matches temporal_boundary=8 during training: 8 * 8 = 64 frames
|
||||
num_frames: 64
|
||||
|
||||
# Negative prompt to avoid unwanted artifacts
|
||||
negative_prompt: "worst quality, inconsistent motion, blurry, jittery, distorted"
|
||||
|
||||
# Output video dimensions [width, height, frames]
|
||||
# Width and height must be divisible by 32
|
||||
# Frames must satisfy: frames % 8 == 1 (e.g., 1, 9, 17, 25, 33, 41, 49, 57, 65, 73, 81, 89, ...)
|
||||
video_dims: [ 576, 576, 89 ]
|
||||
|
||||
# Frame rate for generated videos
|
||||
frame_rate: 25.0
|
||||
|
||||
# Random seed for reproducible validation outputs
|
||||
seed: 42
|
||||
|
||||
# Number of denoising steps for validation inference
|
||||
# Higher values = better quality but slower generation
|
||||
inference_steps: 30
|
||||
|
||||
# Generate validation videos every N training steps
|
||||
# Set to null to disable validation during training
|
||||
interval: 100
|
||||
|
||||
# Classifier-free guidance scale
|
||||
# Higher values = stronger adherence to prompt but may introduce artifacts
|
||||
guidance_scale: 4.0
|
||||
|
||||
# STG (Spatio-Temporal Guidance) parameters for improved video quality
|
||||
# STG is combined with CFG for better temporal coherence
|
||||
stg_scale: 1.0 # Recommended: 1.0 (0.0 disables STG)
|
||||
stg_blocks: [29] # Recommended: single block 29
|
||||
stg_mode: "stg_av" # "stg_av" perturbs both audio and video, "stg_v" video only
|
||||
|
||||
# Whether to generate audio in validation samples
|
||||
# Independent of training_strategy.audio.is_generated - you can generate audio
|
||||
# in validation even when not training the audio branch
|
||||
generate_audio: true
|
||||
|
||||
# Skip validation at the beginning of training (step 0)
|
||||
skip_initial_validation: false
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Checkpoint Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Controls model checkpoint saving during training.
|
||||
checkpoints:
|
||||
# Save a checkpoint every N steps
|
||||
# Set to null to disable intermediate checkpoints
|
||||
interval: 250
|
||||
|
||||
# Number of most recent checkpoints to keep
|
||||
# Set to -1 to keep all checkpoints
|
||||
keep_last_n: -1
|
||||
|
||||
# Precision to use when saving checkpoint weights
|
||||
# Options: "bfloat16" (default, smaller files) or "float32" (full precision)
|
||||
precision: "bfloat16"
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Flow Matching Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Parameters for the flow matching training objective.
|
||||
flow_matching:
|
||||
# Timestep sampling mode
|
||||
# "shifted_logit_normal" is recommended for LTX-2 models
|
||||
timestep_sampling_mode: "shifted_logit_normal"
|
||||
|
||||
# Additional parameters for timestep sampling
|
||||
timestep_sampling_params: { }
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hugging Face Hub Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for uploading trained models to the Hugging Face Hub.
|
||||
hub:
|
||||
# Whether to push the trained model to the Hub
|
||||
push_to_hub: false
|
||||
|
||||
# Repository ID on Hugging Face Hub (e.g., "username/my-lora-model")
|
||||
# Required if push_to_hub is true
|
||||
hub_model_id: null
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Weights & Biases Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Settings for experiment tracking with W&B.
|
||||
wandb:
|
||||
# Enable W&B logging
|
||||
enabled: false
|
||||
|
||||
# W&B project name
|
||||
project: "ltx-2-trainer"
|
||||
|
||||
# W&B username or team (null uses default account)
|
||||
entity: null
|
||||
|
||||
# Tags to help organize runs
|
||||
tags: [ "ltx2", "lora", "video-suffix", "backward-extension" ]
|
||||
|
||||
# Log validation media (video/audio) to W&B
|
||||
log_validation_videos: true
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# General Configuration
|
||||
# -----------------------------------------------------------------------------
|
||||
# Global settings for the training run.
|
||||
|
||||
# Random seed for reproducibility
|
||||
seed: 42
|
||||
|
||||
# Directory to save outputs (checkpoints, validation videos, logs)
|
||||
output_dir: "outputs/video_suffix_lora"
|
||||
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "ltx-trainer"
|
||||
version = "1.1.6"
|
||||
version = "v1.1.6"
|
||||
description = "LTX-2 training, democratized."
|
||||
readme = "README.md"
|
||||
authors = [
|
||||
@@ -54,7 +54,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
|
||||
[tool.ruff]
|
||||
target-version = "1.1.6"
|
||||
target-version = "v1.1.6"
|
||||
line-length = 120
|
||||
# Restrict isort first-party detection to src/ so stray dirs (e.g. wandb/ run output)
|
||||
# next to pyproject.toml don't get classified as first-party packages. See ruff#10519.
|
||||
|
||||
Reference in New Issue
Block a user