acceleration: load_text_encoder_in_8bit: false mixed_precision_mode: bf16 quantization: int8-quanto checkpoints: interval: 200 keep_last_n: 3 precision: bfloat16 data: num_dataloader_workers: 2 preprocessed_data_root: /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/.precomputed_talkvid_v2.2 flow_matching: timestep_sampling_mode: shifted_logit_normal timestep_sampling_params: {} hub: hub_model_id: null push_to_hub: false lora: alpha: 128 dropout: 0.0 rank: 128 target_modules: - audio_attn1.to_k - audio_attn1.to_q - audio_attn1.to_v - audio_attn1.to_out.0 - audio_attn2.to_k - audio_attn2.to_q - audio_attn2.to_v - audio_attn2.to_out.0 - video_to_audio_attn.to_k - video_to_audio_attn.to_q - video_to_audio_attn.to_v - video_to_audio_attn.to_out.0 - audio_to_video_attn.to_k - audio_to_video_attn.to_q - audio_to_video_attn.to_v - audio_to_video_attn.to_out.0 - audio_ff.net.0.proj - audio_ff.net.2 model: load_checkpoint: null model_path: /fast/aviad/github/LTX-2/model/ltx-2-19b-dev.safetensors text_encoder_path: /scratch/aviad/cache/hub/models--google--gemma-3-12b-it-qat-q4_0-unquantized/snapshots/68f7ee4fbd59087436ada77ed2d62f373fdd4482 training_mode: lora optimization: batch_size: 1 enable_gradient_checkpointing: true gradient_accumulation_steps: 4 learning_rate: 0.0002 max_grad_norm: 1.0 optimizer_type: adamw8bit scheduler_params: {} scheduler_type: linear steps: 6000 output_dir: /fast/aviad/github/LTX-2/experiments/week4_training/outputs/2026-02-21/talkvid_audio_ref_only_ic_negpos_r128_v2.2_6k seed: 42 training_strategy: audio_latents_dir: audio_latents first_frame_conditioning_p: 0.9 mask_cross_attention_to_reference: true mask_reference_from_text_attention: true name: audio_ref_only_ic reference_audio_latents_dir: reference_audio_latents use_negative_ref_positions: true validation: captions: null clap_metric: null compute_metrics: true face_metric: arcface frame_rate: 25.0 generate_audio: true guidance_scale: 4.0 images: - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/first_frames/1902__RGaTNCiGwCE_352.100_370.200_seg2.png - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/first_frames/5466__Xhf8qoj1-XI_803.167_813.233_seg0.png - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/first_frames/4095__DuQ7dJIYIhg_19.460_35.420_seg0.png - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/first_frames/1158__7l2KDnCL3Hs_305.956_314.881_seg0.png include_reference_in_output: true inference_steps: 30 interval: 200 mask_cross_attention_to_reference: false mask_ref_audio_to_text: false negative_prompt: worst quality, inconsistent motion, blurry, jittery, distorted prompts: - '[VISUAL]: A medium close-up shot features a woman with long, wavy, reddish-brown hair. She is wearing a dark green, fuzzy sweater and has on noticeable makeup, particularly shimmery eyeshadow. She appears to be speaking directly to the camera with a serious or focused expression, looking slightly down as if reading or focusing on something just below the frame. The background is an indoor setting, possibly a home office or living area. To the left, there''s white shelving with various products and a metallic silver object mounted on the wall, possibly a plaque or decoration. In the center background, there is a light-colored doorway leading into another room. To the right, there is a white armchair with a small, dark-colored dog (possibly a Yorkshire Terrier or similar breed) resting on it. Above the chair hangs a framed black-and-white photograph featuring a classic Mini car in a European city street setting. The lighting is bright and even. [SPEECH]: really designed for lips back in the day when I used it what we would use it for [SOUNDS]: None [TEXT]: None' - '[VISUAL]: A woman with long, wavy, reddish-brown hair is sitting indoors, likely in a bedroom, holding a dark-colored, ornate hardcover book. She is wearing a chunky knit sweater that is predominantly green and white with some red accents, possibly a holiday sweater. She is looking slightly up and to her left while speaking. The background features a collection of Funko Pop figures on shelves to the left, a gold Mickey Mouse figurine hanging on the wall, several decorative Minnie Mouse ear headbands with glittery or festive decorations (some green, some red/white), and a partial view of a decorated Christmas tree to the far right. The wall behind her is a muted reddish-brown color. [SPEECH]: look because that come out in like 2022. [SOUNDS]: The speaker has a moderate, even volume and a calm, conversational tone, speaking in a relaxed manner close to the microphone. [TEXT]: NIGHT SUSAN DENNARD (on the book cover)' - '[VISUAL]: A woman with long, dark brown hair, wearing a dark blue v-neck top, is centered in the frame against a plain white background. She is looking directly at the camera with a focused expression. Her right hand is raised with the index finger pointing upward, and her left hand is slightly clenched near her chest in the beginning of the clip. In the following moments, her hands move to gesture outwards, palms slightly open, and then come together in front of her chest as if in a prayer position or emphasizing a point, before moving outward again with open hands. In the background, to the right, there is a glimpse of a white wall, a small window or opening with a dark frame, and a white doorframe. [SPEECH]: And I also advise on how to make the transferable in real life. [SOUNDS]: The speaker has a calm, even tone and a conversational, slightly formal manner, speaking close to the microphone at a moderate volume. [TEXT]: None' - '[VISUAL]: A young East Asian woman with long, wavy, reddish-brown hair is seated indoors, looking directly at the camera. She is wearing clear, square-framed eyeglasses, a white camisole top with black lace trim, and several pieces of gold jewelry, including necklaces with small charms (one star-shaped, one heart-shaped pendant) and rings. She is gesturing with her hands near her chest, which suggests she is speaking enthusiastically. Her fingernails are painted, some with small red or white dots. The background features a brightly lit room with large windows showing an outdoor scene. To the left, there is a black, beanbag-style chair and a framed piece of abstract art with red and purple hues. To the right, there is a wooden stool with some small items and a faint sign reading "BROADWAY." There''s also a large, white and pink object behind her. [SPEECH]: have prescription I wear contacts and eye exams and Warby Parker glasses start at 95 [SOUNDS]: The speaker has a moderate, conversational tone, sounding calm and direct. They are close to the microphone. [TEXT]: WARBY PARKER, QR code in the upper right corner.' reference_videos: - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2_clean/1902__RGaTNCiGwCE_352.100_370.200_seg1.mp4 - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2_clean/5466__Xhf8qoj1-XI_803.167_813.233_seg1.mp4 - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2_clean/4095__XseXt1xzKr0_296.300_308.780_seg1.mp4 - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2_clean/1158__ye_YGv_jgUc_1375.057_1382.381_seg0.mp4 seed: 42 skip_initial_validation: false speaker_metric: wavlm_ecapa stg_blocks: - 29 stg_mode: stg_av stg_scale: 1.0 target_videos: - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2/1902__RGaTNCiGwCE_352.100_370.200_seg2.mp4 - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2/5466__Xhf8qoj1-XI_803.167_813.233_seg0.mp4 - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2/4095__DuQ7dJIYIhg_19.460_35.420_seg0.mp4 - /fast/aviad/github/LTX-2/datasets/hf/celebv-hq-step9/archives/talkvid_batch_2/1158__7l2KDnCL3Hs_305.956_314.881_seg0.mp4 video_dims: !!python/tuple - 512 - 288 - 121 videos_per_prompt: 1 wandb: enabled: true entity: null log_validation_videos: true project: av-ic-lora tags: - ltx2 - audio-ref-only-ic - r128 - talkvid - talkvid_v2.2 - vad_filtered - max_targets - negative_pos - masked_xattn - text_audio_mask - week4 - 25fps - i2v - rich_captions - validation_no_masks - cross_video_fix - 6k_steps