Views
No views yet
0.4.11adapter: lora
2base_model: dltjdgh0928/test_instruction
3bf16: auto
4chat_template: llama3
5cosine_min_lr_ratio: 0.1
6data_processes: 4
7dataset_prepared_path: null
8datasets:
9- data_files:
10 - fa8b806171f15648_train_data.json
11 ds_type: json
12 format: custom
13 num_proc: 4
14 path: /workspace/input_data/fa8b806171f15648_train_data.json
15 streaming: true
16 type:
17 field_input: table_names
18 field_instruction: question
19 field_output: sql_query
20 format: '{instruction} {input}'
21 no_input_format: '{instruction}'
22 system_format: '{system}'
23 system_prompt: ''
24debug: null
25deepspeed: null
26device_map: sequential
27do_eval: true
28early_stopping_patience: 1
29eval_batch_size: 1
30eval_sample_packing: false
31eval_steps: 25
32evaluation_strategy: steps
33flash_attention: false
34fp16: null
35fsdp: null
36fsdp_config: null
37gradient_accumulation_steps: 16
38gradient_checkpointing: true
39group_by_length: true
40hub_model_id: dada22231/b1153df4-08fb-42de-9f52-127f938dbe79
41hub_strategy: checkpoint
42hub_token: null
43learning_rate: 0.0001
44load_in_4bit: false
45load_in_8bit: false
46local_rank: null
47logging_steps: 1
48lora_alpha: 64
49lora_dropout: 0.05
50lora_fan_in_fan_out: null
51lora_model_dir: null
52lora_r: 32
53lora_target_linear: true
54lora_target_modules:
55- q_proj
56- v_proj
57lr_scheduler: cosine
58max_grad_norm: 1.0
59max_memory:
60 0: 75GB
61 1: 75GB
62 2: 75GB
63 3: 75GB
64 cpu: 96GB
65max_steps: 50
66micro_batch_size: 2
67mixed_precision: bf16
68mlflow_experiment_name: /tmp/fa8b806171f15648_train_data.json
69model_type: AutoModelForCausalLM
70num_epochs: 3
71optim_args:
72 adam_beta1: 0.9
73 adam_beta2: 0.95
74 adam_epsilon: 1e-5
75optimizer: adamw_torch
76output_dir: miner_id_24
77pad_to_sequence_len: true
78resume_from_checkpoint: null
79s2_attention: null
80sample_packing: false
81save_steps: 25
82save_strategy: steps
83sequence_len: 2048
84strict: false
85tf32: false
86tokenizer_type: AutoTokenizer
87torch_compile: false
88torch_dtype: bfloat16
89train_on_inputs: false
90trust_remote_code: true
91use_cache: false
92val_set_size: 50
93wandb_entity: null
94wandb_mode: online
95wandb_name: b1153df4-08fb-42de-9f52-127f938dbe79
96wandb_project: Public_TuningSN
97wandb_runid: b1153df4-08fb-42de-9f52-127f938dbe79
98warmup_ratio: 0.05
99weight_decay: 0.01
100xformers_attention: null
101| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 0.0 | 0.0013 | 1 | nan |
| 0.0 | 0.0322 | 25 | nan |
| 0.0 | 0.0644 | 50 | nan |