project: seed: 12345 resolution: [512, 512] output_dir: artifacts/phase2_balanced_4000_8state paths: comfyui: vendor/ComfyUI ideogram4: vendor/ideogram4 stock_text_encoder: checkpoints/text_encoders/qwen3vl_8b_fp8_scaled.safetensors conditional: checkpoints/diffusion_models/ideogram4_fp8_scaled.safetensors teacher_pairs: artifacts/civitai_balanced_teacher_generation/nemotron_3_ultra_nvfp4_v1/teacher_pairs_4000.jsonl holdout_approval: data/civitai_balanced/teacher_v1/teacher_pairs_sfw_r_x_xxx_4000.approval.json training_index: artifacts/phase2_balanced_4000_training_dataset/index.jsonl teacher: provider: fireworks base_url: https://api.fireworks.ai/inference/v1 model: accounts/fireworks/models/nemotron-3-ultra-nvfp4 temperature: 0.0 max_tokens: 16384 reasoning_effort: high aspect_ratio_placeholder: "1:1" strip_bboxes: true conditioning: dtype: float16 taps: [0, 3, 6, 9, 12, 15, 18, 21, 24, 27, 30, 33, 35] cache_dir: artifacts/phase2_balanced_4000_teacher_conditioning trajectory: # Full-trajectory contract: eight evenly distributed pre-step states from # V4_DEFAULT_20. The computed capture steps are [19, 16, 14, 11, 8, 5, 3, 0]. # Cache manifests are checked before model load, so one-state data is rejected. state_mode: trajectory states_per_prompt: 8 sampling_steps: 20 sampler_preset: V4_DEFAULT_20 seed: 12345 cache_dir: artifacts/phase2_balanced_4000_teacher_trajectories_8state_debanner_s020_v1 generation_batch_size: 1 parallel_cfg: true correction: tensor: checkpoints/debanner/ideogram4_correction_v1.safetensors metadata: checkpoints/debanner/ideogram4_correction_v1.json # Same authoritative direction and operation, explicitly downscaled for # distillation after the 0.6 target produced an excessive first-state bias. strength: 0.2 training: # Adapter geometry is invariant across all production runs: rank/alpha 64, # all seven projection families, and all 36 Qwen blocks (inclusive range). lora_rank: 64 lora_alpha: 64 qwen_blocks: [0, 35] deployment_quantization: scaled_fp8_ste batch_size: 2 gradient_accumulation: 8 adaptive_behavior_batching: enabled: true max_padded_student_tokens: 1017 first_oom_student_tokens: 1223 benchmark_dir: artifacts/phase2_microbatch_cutoff optimizer: adamw optimizer_state_offload: cpu prune_unused_multimodal: true learning_rate: 0.00032 weight_decay: 0.01 default_session_steps: 50 lr_control: mode: windowed_descent_rollback factor: 0.5 minimum_learning_rate: 0.000001 window_steps: 50 probe_every_steps: 10 confirmation_steps: 20 minimum_descent_to_noise: 1.0 minimum_relative_descent: 0.0 validation_every: 50 validation_states_per_record: 1 gradient_clip_norm: 1.0 lambda_velocity: 1.0 lambda_states: 0.0 checkpointing: directory: artifacts/phase2_balanced_4000_8state_debanner_s020_v1_checkpoints every_optimizer_steps: 50 save_first_optimizer_step: true save_on_interrupt: true resume: auto optimizer_boundary_only: true