Views
No views yet
0.4.11adapter: lora
2base_model: microsoft/Phi-3-mini-128k-instruct
3bf16: true
4chat_template: llama3
5data_processes: 54
6dataset_prepared_path: null
7datasets:
8- data_files:
9 - c162e51c5b4c99d0_train_data.json
10 ds_type: json
11 format: custom
12 path: /workspace/input_data/c162e51c5b4c99d0_train_data.json
13 type:
14 field_input: input
15 field_instruction: instruction
16 field_output: output
17 format: '{instruction} {input}'
18 no_input_format: '{instruction}'
19 system_format: '{system}'
20 system_prompt: ''
21debug: null
22deepspeed: null
23device_map: auto
24distributed_training:
25 backend: nccl
26 multi_gpu: true
27 num_gpus: 2
28do_eval: true
29early_stopping_patience: 4
30eval_batch_size: 16
31eval_max_new_tokens: 128
32eval_steps: 150
33eval_table_size: null
34evals_per_epoch: null
35flash_attention: true
36fp16: false
37fsdp:
38- full_shard
39fsdp_config:
40 backward_prefetch: BACKWARD_PRE
41 cpu_offload: false
42 forward_prefetch: false
43 mixed_precision: bf16
44 sharding_strategy: FULL_SHARD
45 use_orig_params: true
46gradient_accumulation_steps: 1
47gradient_checkpointing: true
48group_by_length: true
49hub_ignore_patterns:
50- README.md
51- config.json
52hub_model_id: cimol/79faed67-3425-4896-96fc-bd905512fd9c
53hub_repo: null
54hub_strategy: end
55hub_token: null
56learning_rate: 0.0001
57load_in_4bit: false
58load_in_8bit: false
59local_rank: null
60logging_steps: 10
61lora_alpha: 128
62lora_dropout: 0.1
63lora_fan_in_fan_out: null
64lora_model_dir: null
65lora_r: 64
66lora_target_linear: true
67lr_scheduler: polynomial
68lr_scheduler_warmup_steps: 75
69max_grad_norm: 0.5
70max_memory:
71 0: 75GB
72 1: 75GB
73max_steps: 1950
74micro_batch_size: 16
75mlflow_experiment_name: /tmp/c162e51c5b4c99d0_train_data.json
76model_type: AutoModelForCausalLM
77num_epochs: 4
78optim_args:
79 adam_beta1: 0.9
80 adam_beta2: 0.95
81 adam_epsilon: 1e-8
82optimizer: adamw_torch
83output_dir: miner_id_24
84pad_to_sequence_len: true
85resume_from_checkpoint: null
86s2_attention: null
87sample_packing: false
88save_steps: 300
89saves_per_epoch: null
90seed: 17333
91sequence_len: 1024
92strict: false
93tf32: true
94tokenizer_type: AutoTokenizer
95total_train_batch_size: 32
96train_batch_size: 32
97train_on_inputs: false
98trust_remote_code: true
99val_set_size: 0.05
100wandb_entity: null
101wandb_mode: online
102wandb_name: 3a017f30-2969-4026-ac3a-b3a6f5d66f55
103wandb_project: Gradients-On-Demand
104wandb_run: your_name
105wandb_runid: 3a017f30-2969-4026-ac3a-b3a6f5d66f55
106warmup_steps: 150
107weight_decay: 0.01
108xformers_attention: null
109| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| No log | 0.0001 | 1 | 0.4837 |
| 0.4516 | 0.0218 | 150 | 0.3232 |
| 0.4309 | 0.0437 | 300 | 0.3157 |
| 0.4403 | 0.0655 | 450 | 0.3123 |
| 0.4072 | 0.0873 | 600 | 0.3098 |
| 0.4547 | 0.1092 | 750 | 0.3088 |
| 0.4375 | 0.1310 | 900 | 0.3061 |
| 0.4078 | 0.1528 | 1050 | 0.3062 |
| 0.4762 | 0.1747 | 1200 | 0.3052 |
| 0.4433 | 0.1965 | 1350 | 0.3030 |
| 0.4269 | 0.2183 | 1500 | 0.3030 |
| 0.4119 | 0.2402 | 1650 | 0.3025 |
| 0.4025 | 0.2620 | 1800 | 0.3015 |
| 0.455 | 0.2838 | 1950 | 0.3018 |