Views
No views yet

model_id with Magpie-Align/Llama-3-8B-Magpie-Align-SFT-v0.2.@article{xu2024magpie,
title={Magpie: Alignment Data Synthesis from Scratch by Prompting Aligned LLMs with Nothing},
author={Zhangchen Xu and Fengqing Jiang and Luyao Niu and Yuntian Deng and Radha Poovendran and Yejin Choi and Bill Yuchen Lin},
year={2024},
eprint={2406.08464},
archivePrefix={arXiv},
primaryClass={cs.CL}
}| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 0.8241 | 0.0024 | 1 | 0.8068 |
| 0.5623 | 0.2007 | 85 | 0.5087 |
| 0.4704 | 0.4014 | 170 | 0.4326 |
| 0.4478 | 0.6020 | 255 | 0.4079 |
| 0.4256 | 0.8027 | 340 | 0.3948 |
| 0.4261 | 1.0034 | 425 | 0.3867 |
| 0.3662 | 1.1844 | 510 | 0.3850 |
| 0.363 | 1.3851 | 595 | 0.3823 |
| 0.357 | 1.5858 | 680 | 0.3813 |
| 0.3677 | 1.7865 | 765 | 0.3813 |
0.4.11base_model: meta-llama/Meta-Llama-3-8B
2model_type: LlamaForCausalLM
3tokenizer_type: AutoTokenizer
4
5load_in_8bit: false
6load_in_4bit: false
7strict: false
8
9datasets:
10 - path: Magpie-Align/Magpie-Reasoning-150K
11 type: sharegpt
12 conversation: llama3
13 - path: Magpie-Align/Magpie-Pro-MT-300K-v0.1
14 type: sharegpt
15 conversation: llama3
16dataset_prepared_path: last_run_prepared
17val_set_size: 0.001
18output_dir: axolotl_out/Llama-3-8B-Magpie-Mix-300KMT-150KR
19
20sequence_len: 8192
21sample_packing: true
22eval_sample_packing: false
23pad_to_sequence_len: true
24
25wandb_project: SynDa
26wandb_entity:
27wandb_watch:
28wandb_name: Llama-3-8B-Magpie-Mix-300KMT-150KR
29wandb_log_model:
30hub_model_id: Magpie-Align/Llama-3-8B-Magpie-Mix-300KMT-150KR
31
32gradient_accumulation_steps: 32
33micro_batch_size: 1
34num_epochs: 2
35optimizer: paged_adamw_8bit
36lr_scheduler: cosine
37learning_rate: 2e-5
38
39train_on_inputs: false
40group_by_length: false
41bf16: auto
42fp16:
43tf32: false
44
45gradient_checkpointing: true
46gradient_checkpointing_kwargs:
47 use_reentrant: false
48early_stopping_patience:
49resume_from_checkpoint:
50logging_steps: 1
51xformers_attention:
52flash_attention: true
53
54warmup_ratio: 0.1
55evals_per_epoch: 5
56eval_table_size:
57saves_per_epoch: 1
58debug:
59deepspeed:
60weight_decay: 0.0
61fsdp:
62fsdp_config:
63special_tokens:
64 pad_token: <|end_of_text|>
65