AdwolfCzar commited on
Commit
7b2ca89
·
verified ·
1 Parent(s): 2bf2319

checkpoint step2000

Browse files
checkpoints/step2000/adapter_config.json ADDED
@@ -0,0 +1,116 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {
4
+ "txtfusion.layerwise_blocks.0.attn.gate": 128,
5
+ "txtfusion.layerwise_blocks.0.attn.wk": 128,
6
+ "txtfusion.layerwise_blocks.0.attn.wo": 128,
7
+ "txtfusion.layerwise_blocks.0.attn.wq": 128,
8
+ "txtfusion.layerwise_blocks.0.attn.wv": 128,
9
+ "txtfusion.layerwise_blocks.0.mlp.down": 128,
10
+ "txtfusion.layerwise_blocks.0.mlp.gate": 128,
11
+ "txtfusion.layerwise_blocks.0.mlp.up": 128,
12
+ "txtfusion.layerwise_blocks.1.attn.gate": 128,
13
+ "txtfusion.layerwise_blocks.1.attn.wk": 128,
14
+ "txtfusion.layerwise_blocks.1.attn.wo": 128,
15
+ "txtfusion.layerwise_blocks.1.attn.wq": 128,
16
+ "txtfusion.layerwise_blocks.1.attn.wv": 128,
17
+ "txtfusion.layerwise_blocks.1.mlp.down": 128,
18
+ "txtfusion.layerwise_blocks.1.mlp.gate": 128,
19
+ "txtfusion.layerwise_blocks.1.mlp.up": 128,
20
+ "txtfusion.refiner_blocks.0.attn.gate": 128,
21
+ "txtfusion.refiner_blocks.0.attn.wk": 128,
22
+ "txtfusion.refiner_blocks.0.attn.wo": 128,
23
+ "txtfusion.refiner_blocks.0.attn.wq": 128,
24
+ "txtfusion.refiner_blocks.0.attn.wv": 128,
25
+ "txtfusion.refiner_blocks.0.mlp.down": 128,
26
+ "txtfusion.refiner_blocks.0.mlp.gate": 128,
27
+ "txtfusion.refiner_blocks.0.mlp.up": 128,
28
+ "txtfusion.refiner_blocks.1.attn.gate": 128,
29
+ "txtfusion.refiner_blocks.1.attn.wk": 128,
30
+ "txtfusion.refiner_blocks.1.attn.wo": 128,
31
+ "txtfusion.refiner_blocks.1.attn.wq": 128,
32
+ "txtfusion.refiner_blocks.1.attn.wv": 128,
33
+ "txtfusion.refiner_blocks.1.mlp.down": 128,
34
+ "txtfusion.refiner_blocks.1.mlp.gate": 128,
35
+ "txtfusion.refiner_blocks.1.mlp.up": 128
36
+ },
37
+ "arrow_config": null,
38
+ "auto_mapping": null,
39
+ "base_model_name_or_path": null,
40
+ "bias": "none",
41
+ "corda_config": null,
42
+ "ensure_weight_tying": false,
43
+ "eva_config": null,
44
+ "exclude_modules": null,
45
+ "fan_in_fan_out": false,
46
+ "inference_mode": false,
47
+ "init_lora_weights": true,
48
+ "layer_replication": null,
49
+ "layers_pattern": null,
50
+ "layers_to_transform": null,
51
+ "loftq_config": {},
52
+ "lora_alpha": 64,
53
+ "lora_bias": false,
54
+ "lora_dropout": 0.0,
55
+ "lora_ga_config": null,
56
+ "megatron_config": null,
57
+ "megatron_core": "megatron.core",
58
+ "modules_to_save": null,
59
+ "monteclora_config": null,
60
+ "peft_type": "LORA",
61
+ "peft_version": "0.20.0",
62
+ "qalora_group_size": 16,
63
+ "r": 64,
64
+ "rank_pattern": {
65
+ "txtfusion.layerwise_blocks.0.attn.gate": 128,
66
+ "txtfusion.layerwise_blocks.0.attn.wk": 128,
67
+ "txtfusion.layerwise_blocks.0.attn.wo": 128,
68
+ "txtfusion.layerwise_blocks.0.attn.wq": 128,
69
+ "txtfusion.layerwise_blocks.0.attn.wv": 128,
70
+ "txtfusion.layerwise_blocks.0.mlp.down": 128,
71
+ "txtfusion.layerwise_blocks.0.mlp.gate": 128,
72
+ "txtfusion.layerwise_blocks.0.mlp.up": 128,
73
+ "txtfusion.layerwise_blocks.1.attn.gate": 128,
74
+ "txtfusion.layerwise_blocks.1.attn.wk": 128,
75
+ "txtfusion.layerwise_blocks.1.attn.wo": 128,
76
+ "txtfusion.layerwise_blocks.1.attn.wq": 128,
77
+ "txtfusion.layerwise_blocks.1.attn.wv": 128,
78
+ "txtfusion.layerwise_blocks.1.mlp.down": 128,
79
+ "txtfusion.layerwise_blocks.1.mlp.gate": 128,
80
+ "txtfusion.layerwise_blocks.1.mlp.up": 128,
81
+ "txtfusion.refiner_blocks.0.attn.gate": 128,
82
+ "txtfusion.refiner_blocks.0.attn.wk": 128,
83
+ "txtfusion.refiner_blocks.0.attn.wo": 128,
84
+ "txtfusion.refiner_blocks.0.attn.wq": 128,
85
+ "txtfusion.refiner_blocks.0.attn.wv": 128,
86
+ "txtfusion.refiner_blocks.0.mlp.down": 128,
87
+ "txtfusion.refiner_blocks.0.mlp.gate": 128,
88
+ "txtfusion.refiner_blocks.0.mlp.up": 128,
89
+ "txtfusion.refiner_blocks.1.attn.gate": 128,
90
+ "txtfusion.refiner_blocks.1.attn.wk": 128,
91
+ "txtfusion.refiner_blocks.1.attn.wo": 128,
92
+ "txtfusion.refiner_blocks.1.attn.wq": 128,
93
+ "txtfusion.refiner_blocks.1.attn.wv": 128,
94
+ "txtfusion.refiner_blocks.1.mlp.down": 128,
95
+ "txtfusion.refiner_blocks.1.mlp.gate": 128,
96
+ "txtfusion.refiner_blocks.1.mlp.up": 128
97
+ },
98
+ "revision": null,
99
+ "target_modules": [
100
+ "wv",
101
+ "down",
102
+ "up",
103
+ "gate",
104
+ "wk",
105
+ "wq",
106
+ "wo"
107
+ ],
108
+ "target_parameters": null,
109
+ "task_type": null,
110
+ "trainable_token_indices": null,
111
+ "use_bdlora": null,
112
+ "use_dora": false,
113
+ "use_qalora": false,
114
+ "use_rslora": false,
115
+ "velora_config": null
116
+ }
checkpoints/step2000/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:435102aacdfaeee5120a0e7bc21138d07d62ea1ee7778b0be75019caa177e9c1
3
+ size 484768712
checkpoints/step2000/train.toml ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # KREA2 edit saga 2026-07-29 — treino principal.
2
+ #
3
+ # Base: receita validada krea2_multiref_grounded (docs/KREA2_MULTIREF_RECEITA.md),
4
+ # aqui com max_refs=1 (todos os datasets são pares A->B).
5
+ # - width_shift + refs a t=0 + LoRA condition-only routada + grounding Qwen3-VL
6
+ # - caption_dropout=0.1 (uncond do CFG; condition_dropout é proibido no grounded)
7
+ # - txtfusion_rank=128 (canal de leitura reforçado — tarefa é semântica: editar)
8
+ # - rank 64 (usuário pediu >32; dezenas de milhares de imagens)
9
+
10
+ output_dir = '/workspace/checkpoints/krea2_edit_saga'
11
+ dataset = 'examples/krea2_edit_saga/dataset.toml'
12
+ epochs = 1000
13
+ max_steps = 22500 # 3 épocas × 30.000 pares / batch efetivo 4
14
+ micro_batch_size_per_gpu = 1
15
+ gradient_accumulation_steps = 4
16
+ pipeline_stages = 1
17
+ gradient_clipping = 1.0
18
+ warmup_steps = 100
19
+ save_every_n_steps = 500
20
+ checkpoint_every_n_minutes = 120
21
+ activation_checkpointing = true
22
+ save_dtype = 'bfloat16'
23
+ caching_batch_size = 1 # OBRIGATÓRIO com multi_ref
24
+ blocks_to_swap = 0 # usuário pediu velocidade; VRAM medida 20,1G com swap 8, cabe sem
25
+
26
+ [model]
27
+ type = 'krea2_multiref_grounded'
28
+ diffusion_model = '/workspace/models/krea2/diffusion_models/krea2_raw_fp8_scaled.safetensors'
29
+ vae = '/workspace/models/krea2/split_files/vae/qwen_image_vae.safetensors'
30
+ text_encoders = [
31
+ {path = '/workspace/models/krea2/text_encoders/qwen3vl_4b_bf16.safetensors', type = 'krea2'},
32
+ ]
33
+ dtype = 'bfloat16'
34
+ diffusion_model_dtype = 'float8'
35
+ timestep_sample_method = 'logit_normal'
36
+ flux_shift = true
37
+
38
+ [krea2_multiref_grounded]
39
+ max_refs = 1
40
+ slot_axis = 'width'
41
+ position_mode = 'width_shift'
42
+ reference_position_offset = 1.0
43
+ reference_timestep = 'zero'
44
+ condition_dropout = 0.0 # proibido no caminho grounded
45
+ condition_only_lora = true
46
+ vl_prompt_style = 'picture_n'
47
+ vl_image_label = 'image'
48
+ vl_image_max_pixels = 147456 # 384² — grounding; cache é o gargalo, medir no smoke
49
+ caption_dropout = 0.1
50
+ txtfusion_rank = 128
51
+
52
+ [adapter]
53
+ type = 'lora'
54
+ rank = 64
55
+ dtype = 'bfloat16'
56
+
57
+ [optimizer]
58
+ type = 'AdamW8bitKahan'
59
+ lr = 1e-4
60
+ betas = [0.9, 0.99]
61
+ weight_decay = 0.01
62
+ stabilize = false