Views
No views yet
0.4.11adapter: lora
2base_model: EleutherAI/gpt-neo-125m
3bf16: true
4chat_template: llama3
5dataset_prepared_path: null
6datasets:
7- data_files:
8 - f1bfae7be46056e8_train_data.json
9 ds_type: json
10 format: custom
11 path: /workspace/input_data/f1bfae7be46056e8_train_data.json
12 type:
13 field_input: intent
14 field_instruction: instruction
15 field_output: response_8b_instruct
16 format: '{instruction} {input}'
17 no_input_format: '{instruction}'
18 system_format: '{system}'
19 system_prompt: ''
20debug: null
21deepspeed: null
22early_stopping_patience: 2
23eval_max_new_tokens: 128
24eval_steps: 100
25eval_table_size: null
26flash_attention: false
27fp16: null
28fsdp: null
29fsdp_config: null
30gradient_accumulation_steps: 8
31gradient_checkpointing: true
32group_by_length: false
33hub_model_id: Alphatao/a2fadffe-74bd-40df-9b59-79a4857925a7
34hub_repo: null
35hub_strategy: checkpoint
36hub_token: null
37learning_rate: 0.0002
38load_best_model_at_end: true
39load_in_4bit: false
40load_in_8bit: false
41local_rank: null
42logging_steps: 1
43lora_alpha: 32
44lora_dropout: 0.05
45lora_fan_in_fan_out: null
46lora_model_dir: null
47lora_r: 16
48lora_target_linear: true
49lora_target_modules:
50- q_proj
51- k_proj
52lr_scheduler: cosine
53max_grad_norm: 1.0
54max_steps: 4140
55micro_batch_size: 4
56mlflow_experiment_name: /tmp/f1bfae7be46056e8_train_data.json
57model_type: AutoModelForCausalLM
58num_epochs: 2
59optimizer: adamw_bnb_8bit
60output_dir: miner_id_24
61pad_to_sequence_len: true
62resume_from_checkpoint: null
63s2_attention: null
64sample_packing: false
65save_steps: 100
66sequence_len: 1024
67special_tokens:
68 pad_token: <|endoftext|>
69strict: false
70tf32: true
71tokenizer_type: AutoTokenizer
72train_on_inputs: false
73trust_remote_code: true
74val_set_size: 0.05
75wandb_entity: null
76wandb_mode: online
77wandb_name: ff228523-b38e-4081-8a47-fba2a3a7734a
78wandb_project: Gradients-On-Demand
79wandb_run: your_name
80wandb_runid: ff228523-b38e-4081-8a47-fba2a3a7734a
81warmup_steps: 10
82weight_decay: 0.0
83xformers_attention: null
84| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 14.944 | 0.0004 | 1 | 1.8463 |
| 13.3083 | 0.0410 | 100 | 1.6581 |
| 12.6572 | 0.0819 | 200 | 1.6218 |
| 12.1718 | 0.1229 | 300 | 1.6031 |
| 13.1425 | 0.1638 | 400 | 1.5900 |
| 11.7675 | 0.2048 | 500 | 1.5797 |
| 12.2515 | 0.2458 | 600 | 1.5709 |
| 12.7135 | 0.2867 | 700 | 1.5630 |
| 12.4054 | 0.3277 | 800 | 1.5568 |
| 11.8353 | 0.3686 | 900 | 1.5502 |
| 12.0527 | 0.4096 | 1000 | 1.5448 |
| 12.2754 | 0.4506 | 1100 | 1.5402 |
| 12.0864 | 0.4915 | 1200 | 1.5360 |
| 12.0309 | 0.5325 | 1300 | 1.5317 |
| 10.92 | 0.5734 | 1400 | 1.5285 |
| 12.7007 | 0.6144 | 1500 | 1.5247 |
| 11.1532 | 0.6554 | 1600 | 1.5218 |
| 12.6976 | 0.6963 | 1700 | 1.5187 |
| 11.9156 | 0.7373 | 1800 | 1.5166 |
| 12.6085 | 0.7782 | 1900 | 1.5139 |
| 12.4087 | 0.8192 | 2000 | 1.5118 |
| 12.1581 | 0.8602 | 2100 | 1.5099 |
| 11.9825 | 0.9011 | 2200 | 1.5079 |
| 11.2843 | 0.9421 | 2300 | 1.5060 |
| 12.8118 | 0.9831 | 2400 | 1.5049 |
| 11.1252 | 1.0240 | 2500 | 1.5034 |
| 10.9378 | 1.0650 | 2600 | 1.5022 |
| 12.2633 | 1.1059 | 2700 | 1.5009 |
| 11.5464 | 1.1469 | 2800 | 1.5001 |
| 11.6295 | 1.1879 | 2900 | 1.4994 |
| 12.1325 | 1.2288 | 3000 | 1.4985 |
| 11.571 | 1.2698 | 3100 | 1.4978 |
| 12.381 | 1.3107 | 3200 | 1.4973 |
| 11.4236 | 1.3517 | 3300 | 1.4968 |
| 12.2288 | 1.3927 | 3400 | 1.4964 |
| 12.2337 | 1.4336 | 3500 | 1.4961 |
| 12.5231 | 1.4746 | 3600 | 1.4958 |
| 11.0633 | 1.5155 | 3700 | 1.4957 |
| 12.0329 | 1.5565 | 3800 | 1.4956 |
| 11.9344 | 1.5975 | 3900 | 1.4956 |
| 12.7316 | 1.6384 | 4000 | 1.4955 |
| 12.2778 | 1.6794 | 4100 | 1.4955 |