Views
No views yet
0.4.11adapter: lora
2base_model: unsloth/Llama-3.2-3B-Instruct
3bf16: true
4chat_template: llama3
5dataset_prepared_path: null
6datasets:
7- data_files:
8 - 2473a580e21e684a_train_data.json
9 ds_type: json
10 format: custom
11 path: /workspace/input_data/2473a580e21e684a_train_data.json
12 type:
13 field_input: ground_knowledge
14 field_instruction: query
15 field_output: hit_knowledge
16 format: '{instruction} {input}'
17 no_input_format: '{instruction}'
18 system_format: '{system}'
19 system_prompt: ''
20debug: null
21deepspeed: null
22do_eval: true
23early_stopping_patience: null
24eval_max_new_tokens: 128
25eval_strategy: steps
26eval_table_size: null
27evals_per_epoch: 4
28flash_attention: false
29fp16: null
30fsdp: null
31fsdp_config: null
32gradient_accumulation_steps: 4
33gradient_checkpointing: true
34group_by_length: false
35hub_model_id: eddysang/9378497c-1c38-44f1-892b-47b7727fe2a4
36hub_repo: null
37hub_strategy: checkpoint
38hub_token: null
39learning_rate: 0.0002
40load_in_4bit: false
41load_in_8bit: false
42local_rank: null
43logging_steps: 5
44lora_alpha: 128
45lora_dropout: 0.1
46lora_fan_in_fan_out: null
47lora_model_dir: null
48lora_r: 64
49lora_target_linear: true
50lora_target_modules:
51- q_proj
52- k_proj
53- v_proj
54- o_proj
55- gate_proj
56- down_proj
57- up_proj
58lr_scheduler: cosine
59max_grad_norm: 1
60max_steps: 100
61micro_batch_size: 8
62mlflow_experiment_name: /tmp/2473a580e21e684a_train_data.json
63model_type: AutoModelForCausalLM
64num_epochs: 3
65optim_args:
66 adam_beta1: 0.9
67 adam_beta2: 0.95
68 adam_epsilon: 2.0e-05
69optimizer: adamw_torch
70output_dir: miner_id_24
71pad_to_sequence_len: true
72resume_from_checkpoint: null
73s2_attention: null
74sample_packing: false
75saves_per_epoch: 4
76sequence_len: 1024
77strict: false
78tf32: false
79tokenizer_type: AutoTokenizer
80train_on_inputs: false
81trust_remote_code: true
82val_set_size: 0.1
83wandb_entity: yaudayah0
84wandb_mode: online
85wandb_name: bba600cc-9959-4989-96cb-bd88b8ecffb8
86wandb_project: Gradients-On-Demand
87wandb_run: your_name
88wandb_runid: bba600cc-9959-4989-96cb-bd88b8ecffb8
89warmup_steps: 20
90weight_decay: 0.02
91xformers_attention: false
92| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| No log | 0.0005 | 1 | nan |
| 0.0 | 0.0044 | 9 | nan |
| 0.0 | 0.0088 | 18 | nan |
| 0.0 | 0.0132 | 27 | nan |
| 0.0 | 0.0176 | 36 | nan |
| 0.0 | 0.0220 | 45 | nan |
| 0.0 | 0.0264 | 54 | nan |
| 0.0 | 0.0308 | 63 | nan |
| 0.0 | 0.0352 | 72 | nan |
| 0.0 | 0.0396 | 81 | nan |
| 0.0 | 0.0440 | 90 | nan |
| 0.0 | 0.0484 | 99 | nan |