Views
No views yet
0.4.11adapter: lora
2base_model: unsloth/gemma-2-2b-it
3bf16: true
4chat_template: llama3
5data_processes: 54
6dataset_prepared_path: null
7datasets:
8- data_files:
9 - 8f45cd632fa1120d_train_data.json
10 ds_type: json
11 format: custom
12 path: /workspace/input_data/8f45cd632fa1120d_train_data.json
13 type:
14 field_input: thinking
15 field_instruction: prompt
16 field_output: answer
17 format: '{instruction} {input}'
18 no_input_format: '{instruction}'
19 system_format: '{system}'
20 system_prompt: ''
21debug: null
22deepspeed: null
23device_map: auto
24distributed_training:
25 multi_gpu: true
26 num_gpus: 2
27do_eval: true
28early_stopping_patience: 3
29eval_batch_size: 16
30eval_max_new_tokens: 128
31eval_steps: 1500
32eval_table_size: null
33evals_per_epoch: null
34flash_attention: true
35fp16: false
36fsdp:
37- full_shard
38fsdp_config:
39 activation_checkpointing: true
40 backward_prefetch: BACKWARD_POST
41 forward_prefetch: FORWARD_POST
42 fsdp_min_num_params: 2000000000
43 fsdp_sync_module_states: false
44 limit_all_gathers: true
45 mixed_precision: bf16
46 sharding_strategy: NO_SHARD
47gradient_accumulation_steps: 1
48gradient_checkpointing: false
49group_by_length: true
50hub_ignore_patterns:
51- README.md
52- config.json
53hub_model_id: cimol/f5ca0978-7348-4e89-a5a8-0753f63260a9
54hub_repo: null
55hub_strategy: end
56hub_token: null
57learning_rate: 0.0002
58load_in_4bit: false
59load_in_8bit: false
60local_rank: null
61logging_steps: 100
62lora_alpha: 128
63lora_dropout: 0.05
64lora_fan_in_fan_out: null
65lora_model_dir: null
66lora_r: 64
67lora_target_linear: true
68lr_scheduler: cosine
69lr_scheduler_warmup_steps: 100
70max_grad_norm: 0.5
71max_memory:
72 0: 75GB
73 1: 75GB
74max_steps: 1500
75micro_batch_size: 16
76mlflow_experiment_name: /tmp/8f45cd632fa1120d_train_data.json
77model_type: AutoModelForCausalLM
78num_epochs: 1
79optim_args:
80 adam_beta1: 0.9
81 adam_beta2: 0.95
82 adam_epsilon: 1e-8
83optimizer: adamw_torch
84output_dir: miner_id_24
85pad_to_sequence_len: true
86resume_from_checkpoint: null
87s2_attention: null
88sample_packing: false
89save_steps: 1500
90saves_per_epoch: null
91seed: 17333
92sequence_len: 1024
93strict: false
94tf32: true
95tokenizer_type: AutoTokenizer
96total_train_batch_size: 32
97train_batch_size: 32
98train_on_inputs: false
99trust_remote_code: true
100val_set_size: 0.05
101wandb_entity: null
102wandb_mode: online
103wandb_name: 1254f0ce-e179-4756-bb5c-7b446f52c4ac
104wandb_project: Gradients-On-Demand
105wandb_run: your_name
106wandb_runid: 1254f0ce-e179-4756-bb5c-7b446f52c4ac
107warmup_steps: 100
108weight_decay: 0.005
109xformers_attention: null
110| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| No log | 0.0021 | 1 | 2.4356 |