Views
No views yet
0.4.11adapter: lora
2base_model: Qwen/Qwen2-1.5B-Instruct
3bf16: true
4chat_template: llama3
5dataloader_num_workers: 8
6dataloader_pin_memory: true
7dataset_prepared_path: null
8datasets:
9- data_files:
10 - 7376a9b16ca78e0e_train_data.json
11 ds_type: json
12 format: custom
13 path: /workspace/input_data/
14 type:
15 field_instruction: instruct
16 field_output: output
17 format: '{instruction}'
18 no_input_format: '{instruction}'
19 system_format: '{system}'
20 system_prompt: ''
21debug: null
22deepspeed: null
23device_map: auto
24dynamic_lora_per_layer: true
25early_stopping_patience: 3
26eval_max_new_tokens: 128
27eval_steps: 100
28eval_table_size: null
29evaluation_strategy: steps
30flash_attention: true
31fp16: false
32fsdp: null
33fsdp_config: null
34gradient_accumulation_steps: 2
35gradient_checkpointing: true
36group_by_length: false
37hub_model_id: JoshMe1/74ef9b35-3518-48d8-9d91-f06849ea9261
38hub_repo: null
39hub_strategy: checkpoint
40hub_token: null
41learning_rate: 5.0e-05
42load_in_4bit: false
43load_in_8bit: true
44local_rank: null
45logging_steps: 10
46lora_alpha: 128
47lora_dropout: 0.1
48lora_fan_in_fan_out: null
49lora_model_dir: null
50lora_r: 64
51lora_target_linear: true
52lr_finder: true
53lr_scheduler: cosine
54lr_scheduler_args: []
55max_grad_norm: 1.0
56max_memory:
57 0: 130GB
58max_steps: 3972
59micro_batch_size: 8
60mixed_precision: bf16
61mlflow_experiment_name: /tmp/7376a9b16ca78e0e_train_data.json
62model_type: AutoModelForCausalLM
63num_epochs: 10
64optimizer: adamw_bnb_8bit
65output_dir: miner_id_24
66pad_to_sequence_len: true
67resume_from_checkpoint: null
68s2_attention: null
69sample_packing: false
70save_steps: 100
71save_strategy: steps
72save_total_limit: 3
73scheduler:
74 factor: 0.5
75 monitor: eval_loss
76 patience: 1
77 threshold: 0.01
78 type: ReduceLROnPlateau
79sequence_len: 512
80strict: false
81tf32: false
82tokenizer_type: AutoTokenizer
83train_on_inputs: false
84training_stages:
85- learning_rate: 5.0e-05
86 name: warmup
87 num_train_epochs: 1
88- learning_rate: 5.0e-06
89 name: main
90trl:
91 ema: true
92 ema_decay: 0.999
93trust_remote_code: true
94val_set_size: 0.05
95wandb_entity: null
96wandb_mode: online
97wandb_name: fe9696d5-94a1-410e-8793-2418f6c2af21
98wandb_project: Gradients-On-Demand
99wandb_run: your_name
100wandb_runid: fe9696d5-94a1-410e-8793-2418f6c2af21
101warmup_steps: 397
102weight_decay: 0.01
103xformers_attention: true
104| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| No log | 0.0011 | 1 | 1.9862 |
| 1.7642 | 0.1059 | 100 | 1.7310 |
| 1.7102 | 0.2119 | 200 | 1.6722 |
| 1.6491 | 0.3178 | 300 | 1.6463 |
| 1.6841 | 0.4237 | 400 | 1.6313 |
| 1.6463 | 0.5297 | 500 | 1.6162 |
| 1.6142 | 0.6356 | 600 | 1.6052 |
| 1.5927 | 0.7415 | 700 | 1.5980 |
| 1.5904 | 0.8475 | 800 | 1.5901 |
| 1.6223 | 0.9534 | 900 | 1.5842 |
| 1.5658 | 1.0593 | 1000 | 1.5817 |
| 1.5308 | 1.1653 | 1100 | 1.5786 |
| 1.526 | 1.2712 | 1200 | 1.5749 |
| 1.5085 | 1.3771 | 1300 | 1.5710 |
| 1.5275 | 1.4831 | 1400 | 1.5673 |
| 1.4933 | 1.5890 | 1500 | 1.5632 |
| 1.4964 | 1.6949 | 1600 | 1.5610 |
| 1.5289 | 1.8008 | 1700 | 1.5571 |
| 1.5264 | 1.9068 | 1800 | 1.5542 |
| 1.4264 | 2.0127 | 1900 | 1.5608 |
| 1.4112 | 2.1186 | 2000 | 1.5626 |
| 1.4097 | 2.2246 | 2100 | 1.5638 |