Views
No views yet
0.4.11adapter: lora
2base_model: echarlaix/tiny-random-mistral
3bf16: true
4chat_template: llama3
5dataset_prepared_path: null
6datasets:
7- data_files:
8 - 6b60a00d2240e919_train_data.json
9 ds_type: json
10 format: custom
11 path: /workspace/input_data/6b60a00d2240e919_train_data.json
12 type:
13 field_instruction: thinking
14 field_output: raw_emails
15 format: '{instruction}'
16 no_input_format: '{instruction}'
17 system_format: '{system}'
18 system_prompt: ''
19debug: null
20device_map:
21 ? ''
22 : 0,1,2,3,4,5,6,7
23early_stopping_patience: 2
24eval_max_new_tokens: 128
25eval_steps: 100
26eval_table_size: null
27flash_attention: true
28gradient_accumulation_steps: 8
29gradient_checkpointing: true
30group_by_length: false
31hub_model_id: Alphatao/54d2a404-778d-4493-872e-39668b00c3d4
32hub_repo: null
33hub_strategy: null
34hub_token: null
35learning_rate: 0.0002
36load_best_model_at_end: true
37load_in_4bit: false
38load_in_8bit: false
39local_rank: null
40logging_steps: 1
41lora_alpha: 32
42lora_dropout: 0.05
43lora_fan_in_fan_out: null
44lora_model_dir: null
45lora_r: 16
46lora_target_linear: true
47lora_target_modules:
48- q_proj
49- k_proj
50- v_proj
51- o_proj
52- down_proj
53- up_proj
54lr_scheduler: cosine
55max_grad_norm: 1.0
56max_steps: 4140
57micro_batch_size: 4
58mlflow_experiment_name: /tmp/6b60a00d2240e919_train_data.json
59model_type: AutoModelForCausalLM
60num_epochs: 2
61optimizer: adamw_bnb_8bit
62output_dir: miner_id_24
63pad_to_sequence_len: true
64resume_from_checkpoint: null
65s2_attention: null
66sample_packing: false
67save_steps: 100
68sequence_len: 2048
69special_tokens:
70 pad_token: </s>
71strict: false
72tf32: true
73tokenizer_type: AutoTokenizer
74train_on_inputs: false
75trust_remote_code: true
76val_set_size: 0.044419569485532544
77wandb_entity: null
78wandb_mode: online
79wandb_name: ecf87c37-5377-4e51-a311-a1485a15b7d2
80wandb_project: Gradients-On-Demand
81wandb_run: your_name
82wandb_runid: ecf87c37-5377-4e51-a311-a1485a15b7d2
83warmup_steps: 10
84weight_decay: 0.0
85xformers_attention: null
86| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 83.033 | 0.0003 | 1 | 10.3781 |
| 82.3494 | 0.0297 | 100 | 10.2917 |
| 82.1363 | 0.0595 | 200 | 10.2662 |
| 81.9565 | 0.0892 | 300 | 10.2412 |
| 81.8297 | 0.1190 | 400 | 10.2265 |
| 81.7878 | 0.1487 | 500 | 10.2145 |
| 81.692 | 0.1785 | 600 | 10.2060 |
| 81.6421 | 0.2082 | 700 | 10.1992 |
| 81.5873 | 0.2380 | 800 | 10.1951 |
| 81.6027 | 0.2677 | 900 | 10.1918 |
| 81.6317 | 0.2975 | 1000 | 10.1892 |
| 81.6427 | 0.3272 | 1100 | 10.1874 |
| 81.5747 | 0.3570 | 1200 | 10.1856 |
| 81.4871 | 0.3867 | 1300 | 10.1843 |
| 81.5444 | 0.4165 | 1400 | 10.1832 |
| 81.52 | 0.4462 | 1500 | 10.1821 |
| 81.505 | 0.4760 | 1600 | 10.1813 |
| 81.5419 | 0.5057 | 1700 | 10.1804 |
| 81.5173 | 0.5355 | 1800 | 10.1797 |
| 81.5497 | 0.5652 | 1900 | 10.1790 |
| 81.5006 | 0.5950 | 2000 | 10.1784 |
| 81.5576 | 0.6247 | 2100 | 10.1777 |
| 81.4592 | 0.6545 | 2200 | 10.1771 |
| 81.442 | 0.6842 | 2300 | 10.1766 |
| 81.5106 | 0.7140 | 2400 | 10.1760 |
| 81.5266 | 0.7437 | 2500 | 10.1754 |
| 81.4714 | 0.7735 | 2600 | 10.1749 |
| 81.4872 | 0.8032 | 2700 | 10.1744 |
| 81.4705 | 0.8330 | 2800 | 10.1742 |
| 81.5065 | 0.8627 | 2900 | 10.1739 |
| 81.4965 | 0.8925 | 3000 | 10.1737 |
| 81.5025 | 0.9222 | 3100 | 10.1735 |
| 81.5072 | 0.9520 | 3200 | 10.1734 |
| 81.5161 | 0.9817 | 3300 | 10.1732 |
| 81.4285 | 1.0115 | 3400 | 10.1731 |
| 81.4996 | 1.0412 | 3500 | 10.1731 |
| 81.4002 | 1.0710 | 3600 | 10.1730 |
| 81.4594 | 1.1007 | 3700 | 10.1730 |
| 81.4385 | 1.1305 | 3800 | 10.1729 |
| 81.4901 | 1.1602 | 3900 | 10.1729 |
| 81.5151 | 1.1900 | 4000 | 10.1729 |
| 81.473 | 1.2197 | 4100 | 10.1729 |