Views
No views yet
0.4.11adapter: lora
2base_model: Maykeye/TinyLLama-v0
3bf16: auto
4chat_template: llama3
5dataset_prepared_path: null
6datasets:
7- format: custom
8 path: samoline/api-generator
9 type:
10 field_instruction: query
11 field_output: functions
12 format: '{instruction}'
13 no_input_format: '{instruction}'
14 system_format: '{system}'
15 system_prompt: ''
16debug: null
17deepspeed: null
18early_stopping_patience: null
19eval_max_new_tokens: 128
20eval_table_size: null
21evals_per_epoch: 4
22flash_attention: false
23fp16: null
24fsdp: null
25fsdp_config: null
26gradient_accumulation_steps: 4
27gradient_checkpointing: false
28group_by_length: false
29hub_model_id: samoline/e3947cb3-7c20-41c7-b215-9a5e96c2917b
30hub_repo: samoline
31hub_strategy: checkpoint
32hub_token: null
33learning_rate: 0.0002
34load_in_4bit: false
35load_in_8bit: false
36local_rank: null
37logging_steps: 1
38lora_alpha: 16
39lora_dropout: 0.05
40lora_fan_in_fan_out: null
41lora_model_dir: null
42lora_r: 8
43lora_target_linear: true
44lr_scheduler: cosine
45max_steps: 10
46micro_batch_size: 2
47mlflow_experiment_name: samoline/api-generator
48model_type: AutoModelForCausalLM
49num_epochs: 1
50optimizer: adamw_bnb_8bit
51output_dir: miner_id_24
52pad_to_sequence_len: true
53resume_from_checkpoint: null
54s2_attention: null
55sample_packing: false
56saves_per_epoch: 4
57sequence_len: 512
58special_tokens:
59 pad_token: </s>
60strict: false
61tf32: false
62tokenizer_type: AutoTokenizer
63train_on_inputs: false
64trust_remote_code: true
65val_set_size: 0.05
66wandb_entity: samoline-nan
67wandb_mode: online
68wandb_name: e3947cb3-7c20-41c7-b215-9a5e96c2917b
69wandb_project: Gradients-On-Demand
70wandb_run: dev
71wandb_runid: e3947cb3-7c20-41c7-b215-9a5e96c2917b
72warmup_steps: 10
73weight_decay: 0.0
74xformers_attention: null
75| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 10.0927 | 0.0010 | 1 | 10.5287 |
| 10.8963 | 0.0031 | 3 | 10.5182 |
| 10.335 | 0.0062 | 6 | 10.3617 |
| 10.3368 | 0.0092 | 9 | 10.0317 |