Views
No views yet
0.4.11adapter: lora
2base_model: NousResearch/Nous-Hermes-2-Mistral-7B-DPO
3bf16: auto
4chat_template: llama3
5dataset_prepared_path: null
6datasets:
7- data_files:
8 - b99507c826e2be9c_train_data.json
9 ds_type: json
10 format: custom
11 path: /workspace/input_data/b99507c826e2be9c_train_data.json
12 type:
13 field_input: evidence
14 field_instruction: question
15 field_output: SQL
16 format: '{instruction} {input}'
17 no_input_format: '{instruction}'
18 system_format: '{system}'
19 system_prompt: ''
20ddp_find_unused_parameters: false
21distributed_type: ddp
22early_stopping_patience: null
23env:
24 CUDA_VISIBLE_DEVICES: 0,1
25 MASTER_ADDR: localhost
26 MASTER_PORT: '29500'
27 NCCL_DEBUG: INFO
28 NCCL_IB_DISABLE: '1'
29 NCCL_P2P_DISABLE: '1'
30 PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True, max_split_size_mb:512, garbage_collection_threshold:0.8
31 WORLD_SIZE: '2'
32eval_max_new_tokens: 128
33eval_table_size: null
34evals_per_epoch: 4
35flash_attention: false
36fp16: null
37gradient_accumulation_steps: 4
38gradient_checkpointing: true
39group_by_length: false
40hub_model_id: fats-fme/62e8d22f-0c7c-426d-aa67-4b20b4d9c2ba
41hub_repo: null
42hub_strategy: checkpoint
43hub_token: null
44learning_rate: 0.0002
45load_in_4bit: false
46load_in_8bit: true
47logging_steps: 1
48lora_alpha: 32
49lora_dropout: 0.05
50lora_fan_in_fan_out: null
51lora_model_dir: null
52lora_r: 16
53lora_target_linear: true
54lr_scheduler: cosine
55max_memory_MB: 35000
56max_steps: 10
57micro_batch_size: 2
58mlflow_experiment_name: /tmp/b99507c826e2be9c_train_data.json
59model_type: AutoModelForCausalLM
60num_devices: 2
61num_epochs: 1
62optimizer: adamw_bnb_8bit
63output_dir: miner_id_24
64pad_to_sequence_len: true
65resume_from_checkpoint: null
66s2_attention: null
67sample_packing: false
68saves_per_epoch: 4
69sequence_len: 4056
70strict: false
71tf32: false
72tokenizer_type: AutoTokenizer
73train_on_inputs: false
74trust_remote_code: true
75val_set_size: 0.05
76wandb_entity: null
77wandb_mode: online
78wandb_name: 62e8d22f-0c7c-426d-aa67-4b20b4d9c2ba
79wandb_project: Gradients-On-Demand
80wandb_run: your_name
81wandb_runid: 62e8d22f-0c7c-426d-aa67-4b20b4d9c2ba
82warmup_steps: 10
83world_size: 2
84xformers_attention: true
85| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 0.0 | 0.0018 | 1 | nan |
| 0.0 | 0.0055 | 3 | nan |
| 0.0 | 0.0109 | 6 | nan |
| 0.0 | 0.0164 | 9 | nan |