Views
No views yet
PocketDoc/Dans-PersonalityEngine-V1.2.0-24b (a jack of all trades instruct model), which was trained ontop of mistralai/Mistral-Small-24B-Base-2501.<think>\n to the start of the assistant turn.

0.8.0.dev01mlflow_tracking_uri: http://127.0.0.1:7860
2mlflow_experiment_name: MS-2501-DPE-QwQify-v0.1-24B-LoRA
3
4# Hugging Face saving config
5hub_model_id: BeaverAI/MS-2501-DPE-QwQify-v0.1-24B-LoRA-WS
6hub_strategy: every_save
7
8# Model checkpointing config
9output_dir: ./Outputs/MS-2501-DPE-QwQify-v0.1-24B-LoRA
10resume_from_checkpoint:
11save_steps: 50
12save_safetensors: true
13save_total_limit: 3
14save_only_model: false
15
16# Model architecture config
17base_model: PocketDoc/Dans-PersonalityEngine-V1.2.0-24b
18model_type: MistralForCausalLM
19tokenizer_type: AutoTokenizer
20
21# Mixed precision training config
22bf16: true
23fp16: false
24tf32: false
25
26# Model loading config
27load_in_8bit: false
28load_in_4bit: false
29strict: false
30
31# Sequence config
32sequence_len: 8192
33min_sample_len: 256
34sample_packing: true
35eval_sample_packing: true
36pad_to_sequence_len: true
37train_on_inputs: false
38group_by_length: false
39
40# LoRA adapter config
41adapter: lora
42lora_model_dir:
43lora_r: 128
44lora_alpha: 128
45lora_dropout: 0.125
46peft_layers_to_transform:
47peft_use_dora:
48peft_use_rslora:
49peft_layer_replication:
50lora_target_modules:
51 - gate_proj
52 - down_proj
53 - up_proj
54 - q_proj
55 - v_proj
56 - k_proj
57 - o_proj
58lora_modules_to_save:
59
60# Fix uninitialized tokens (such as <|start_header_id|> on the base L3 models)
61fix_untrained_tokens:
62
63# Dataset config
64# https://github.com/xzuyn/axolotl/blob/came-plus-formatters/src/axolotl/prompt_strategies/customchatml-regex-last-only.py
65datasets:
66 - path: PJMixers-Dev/allura-org_gryphe-sonnet-3.5-charcards-names-added-qwq-all-aphrodite-Shuffled
67 split: train
68 type: customchatml-regex-last-only
69 - path: PJMixers-Dev/anthracite-org_c2_logs_32k_llama3_qwen2_v1.3-qwq-all-aphrodite-Shuffled
70 split: train
71 type: customchatml-regex-last-only
72 - path: PJMixers-Dev/grimulkan_aicg-logs-augmented-system-qwq-all-aphrodite-Shuffled
73 split: train
74 type: customchatml-regex-last-only
75 - path: PJMixers-Dev/grimulkan_jannie-log-augmented-system-qwq-all-aphrodite-Shuffled
76 split: train
77 type: customchatml-regex-last-only
78 - path: PJMixers-Dev/grimulkan_PIPPA-augmented-dedup-system-qwq-all-aphrodite-Shuffled
79 split: train
80 type: customchatml-regex-last-only
81 - path: PJMixers-Dev/lemonilia_LimaRP-Only-NonSus-Simple-CustomShareGPT-qwq-all-aphrodite-Shuffled
82 split: train
83 type: customchatml-regex-last-only
84 - path: PJMixers-Dev/MinervaAI_Aesir-Preview-Anon-qwq-all-aphrodite-Shuffled
85 split: train
86 type: customchatml-regex-last-only
87 - path: PJMixers-Dev/NyxKrage_chub-logs-sharegpt-longest-CustomShareGPT-qwq-all-aphrodite-Shuffled
88 split: train
89 type: customchatml-regex-last-only
90 - path: PJMixers-Dev/PocketDoc_Dans-Prosemaxx-Cowriter-XL-8192-shrunk-l3-qwq-all-aphrodite-Shuffled
91 split: train
92 type: customchatml-regex-last-only
93 - path: PJMixers-Dev/PocketDoc_Dans-Personamaxx-Rainy-qwq-all-aphrodite-Shuffled
94 split: train
95 type: customchatml-regex-last-only
96test_datasets:
97 - path: PJMixers-Dev/allura-org_gryphe-sonnet-3.5-charcards-names-added-qwq-all-aphrodite-Shuffled
98 split: test
99 type: customchatml-regex-last-only
100 - path: PJMixers-Dev/anthracite-org_c2_logs_32k_llama3_qwen2_v1.3-qwq-all-aphrodite-Shuffled
101 split: test
102 type: customchatml-regex-last-only
103 - path: PJMixers-Dev/grimulkan_aicg-logs-augmented-system-qwq-all-aphrodite-Shuffled
104 split: test
105 type: customchatml-regex-last-only
106 - path: PJMixers-Dev/grimulkan_jannie-log-augmented-system-qwq-all-aphrodite-Shuffled
107 split: test
108 type: customchatml-regex-last-only
109 - path: PJMixers-Dev/grimulkan_PIPPA-augmented-dedup-system-qwq-all-aphrodite-Shuffled
110 split: test
111 type: customchatml-regex-last-only
112 - path: PJMixers-Dev/lemonilia_LimaRP-Only-NonSus-Simple-CustomShareGPT-qwq-all-aphrodite-Shuffled
113 split: test
114 type: customchatml-regex-last-only
115 - path: PJMixers-Dev/MinervaAI_Aesir-Preview-Anon-qwq-all-aphrodite-Shuffled
116 split: test
117 type: customchatml-regex-last-only
118 - path: PJMixers-Dev/NyxKrage_chub-logs-sharegpt-longest-CustomShareGPT-qwq-all-aphrodite-Shuffled
119 split: test
120 type: customchatml-regex-last-only
121 - path: PJMixers-Dev/PocketDoc_Dans-Prosemaxx-Cowriter-XL-8192-shrunk-l3-qwq-all-aphrodite-Shuffled
122 split: test
123 type: customchatml-regex-last-only
124 - path: PJMixers-Dev/PocketDoc_Dans-Personamaxx-Rainy-qwq-all-aphrodite-Shuffled
125 split: test
126 type: customchatml-regex-last-only
127val_set_size: 0
128eval_strategy: steps
129eval_steps: 50
130dataset_prepared_path: ./00-Tokenized-Datasets/MS-2501-DPE-QwQify-v0.1-24B-customchatml-regex-last-only
131shuffle_merged_datasets: true
132dataset_processes:
133
134# Training hyperparameters
135num_epochs: 2
136gradient_accumulation_steps: 1
137micro_batch_size: 8 # x4 GPUs = 32
138eval_batch_size: 8 # x4 GPUs = 32
139warmup_steps: 0
140optimizer: came_pytorch
141optim_args:
142optim_target_modules:
143lr_scheduler: rex
144learning_rate: 2e-5
145cosine_min_lr_ratio:
146loraplus_lr_ratio:
147loraplus_lr_embedding:
148weight_decay: 0.1
149max_grad_norm: 1
150logging_steps: 1
151
152# Model optimization
153gradient_checkpointing: unsloth
154flash_attention: true
155plugins:
156 - axolotl.integrations.liger.LigerPlugin
157 - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
158cut_cross_entropy: true
159liger_rope: true
160liger_rms_norm: true
161liger_layer_norm: true
162liger_glu_activation: true
163liger_cross_entropy: false
164liger_fused_linear_cross_entropy: false
165lora_mlp_kernel: false
166lora_qkv_kernel: false
167lora_o_kernel: false
168
169# DeepSpeed
170deepspeed: deepspeed_configs/zero3_bf16.json
171
172# Garbage Collection
173gc_steps: 1
174
175# Debug config
176debug: true
177seed: 42
178
179# Token config
180special_tokens:
181 bos_token: "<s>"
182 eos_token: "<|im_end|>"
183 pad_token: "<pad>"
184tokens:
185| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 1.9925 | 0.0019 | 1 | 1.9225 |
| 1.4228 | 0.0936 | 50 | 1.4329 |
| 1.3473 | 0.1873 | 100 | 1.3722 |
| 1.3259 | 0.2809 | 150 | 1.3414 |
| 1.2795 | 0.3745 | 200 | 1.3199 |
| 1.2817 | 0.4682 | 250 | 1.3029 |
| 1.2365 | 0.5618 | 300 | 1.2910 |
| 1.2134 | 0.6554 | 350 | 1.2803 |
| 1.2655 | 0.7491 | 400 | 1.2700 |
| 1.2297 | 0.8427 | 450 | 1.2614 |
| 1.178 | 0.9363 | 500 | 1.2524 |
| 1.1525 | 1.0300 | 550 | 1.2467 |
| 1.1751 | 1.1236 | 600 | 1.2411 |
| 1.216 | 1.2172 | 650 | 1.2366 |
| 1.1706 | 1.3109 | 700 | 1.2302 |
| 1.1363 | 1.4045 | 750 | 1.2256 |
| 1.1563 | 1.4981 | 800 | 1.2194 |
| 1.1559 | 1.5918 | 850 | 1.2147 |
| 1.1263 | 1.6854 | 900 | 1.2090 |
| 1.099 | 1.7790 | 950 | 1.2038 |
| 1.1786 | 1.8727 | 1000 | 1.1994 |
| 1.1057 | 1.9663 | 1050 | 1.1949 |