Views
No views yet
0.10.0.dev01adapter: null
2base_model: /cache/models/Qwen--Qwen2.5-0.5B-Instruct
3bf16: true
4chat_template: llama3
5cosine_min_lr_ratio: 0.3
6dataset_prepared_path: null
7datasets:
8- data_files:
9 - /workspace/axolotl/data/746e620a-82c7-4581-a764-7685a2eaffb6_train_data.json
10 ds_type: json
11 format: custom
12 path: /workspace/axolotl/data
13 type:
14 field_instruction: instruct
15 field_output: output
16 format: '{instruction}'
17 no_input_format: '{instruction}'
18 system_format: '{system}'
19 system_prompt: ''
20ddp: true
21debug: null
22deepspeed: null
23early_stopping_patience: null
24eval_sample_packing: false
25eval_table_size: null
26evals_per_epoch: 1
27flash_attention: true
28fp16: null
29fsdp: null
30fsdp_config: null
31gradient_accumulation_steps: 4
32gradient_checkpointing: true
33group_by_length: false
34hub_model_id: skrd3/dc051df3-c873-4e46-ae09-291fb24b8c3a
35hub_repo: null
36hub_strategy: checkpoint
37hub_token: null
38learning_rate: 0.0001
39liger_fused_linear_cross_entropy: true
40liger_glu_activation: true
41liger_layer_norm: true
42liger_rms_norm: true
43liger_rope: true
44load_in_4bit: false
45load_in_8bit: false
46local_rank: null
47logging_steps: 1
48lora_alpha: 512
49lora_dropout: 0.1
50lora_fan_in_fan_out: null
51lora_model_dir: null
52lora_r: 128
53lora_target_linear: true
54lr_scheduler: cosine
55micro_batch_size: 15
56mlflow_experiment_name: /workspace/axolotl/data/746e620a-82c7-4581-a764-7685a2eaffb6_train_data.json
57model_type: AutoModelForCausalLM
58num_epochs: 3
59optimizer: paged_adamw_8bit
60output_dir: /app/checkpoints/746e620a-82c7-4581-a764-7685a2eaffb6/temp_trainer
61pad_to_sequence_len: true
62plugins:
63- axolotl.integrations.liger.LigerPlugin
64resume_from_checkpoint: null
65s2_attention: null
66sample_packing: true
67save_only_model: true
68saves_per_epoch: 1
69sequence_len: 1024
70strict: false
71tf32: false
72tokenizer_type: AutoTokenizer
73train_on_inputs: false
74trust_remote_code: true
75use_liger: true
76val_set_size: 0.03278151122766759
77wandb_entity: null
78wandb_mode: online
79wandb_project: Gradients-On-Demand
80wandb_run: your_name
81wandb_runid: default
82warmup_steps: 10
83weight_decay: 0.0
84xformers_attention: null
85| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 0.947 | 0.0563 | 1 | 1.0290 |
| 0.5193 | 0.9577 | 17 | 0.5769 |
| 0.3624 | 1.9577 | 34 | 0.5381 |
| 0.2263 | 2.9577 | 51 | 0.5795 |