Views
No views yet
1training:
2 max_seq_length: 4096
3 num_epochs: 3
4 learning_rate: 0.0002
5 batch_size: 2
6 gradient_accumulation_steps: 4
7 warmup_steps: 10
8 max_steps: 0
9 save_steps: 200
10 eval_steps: 0
11 weight_decay: 0.001
12 random_seed: 3407
13 packing: false
14 train_on_completions: true
15 gradient_checkpointing: unsloth
16 optim: adamw_8bit
17 lr_scheduler_type: linear
18lora:
19 lora_r: 32
20 lora_alpha: 64
21 lora_dropout: 0.05
22 target_modules:
23 - q_proj
24 - k_proj
25 - v_proj
26 - o_proj
27 - gate_proj
28 - up_proj
29 - down_proj
30 use_rslora: false
31 use_loftq: false
32 finetune_vision_layers: false