Views
No views yet

Esta es una versión preliminar del dataset card. El modelo está en desarrollo y no es la versión final. Si quieres saber más sobre este modelo, escribe a iker.garciaf@ehu.eus

1base_model: meta-llama/Meta-Llama-3-8B-Instruct
2model_type: AutoModelForCausalLM
3tokenizer_type: AutoTokenizer
4is_falcon_derived_model:
5is_llama_derived_model:
6is_qwen_derived_model:
7is_mistral_derived_model:
8
9load_in_8bit: false
10load_in_4bit: false
11strict: false
12
13device_map: null
14
15datasets:
16 - path: /ikerlariak/igarcia945/InstructDatasets/Barcenas-Economia.jsonl
17 type: sharegpt
18 conversation: llama3
19 field: conversations
20 roles:
21 input:
22 - system
23 - gpt
24 output:
25 - human
26 - path: /ikerlariak/igarcia945/InstructDatasets/casimedicos.jsonl
27 type: sharegpt
28 conversation: llama3
29 field: conversations
30 roles:
31 input:
32 - system
33 - gpt
34 output:
35 - human
36 - path: /ikerlariak/igarcia945/InstructDatasets/coser_resumene.jsonl
37 type: sharegpt
38 conversation: llama3
39 field: conversations
40 roles:
41 input:
42 - system
43 - gpt
44 output:
45 - human
46 - path: /ikerlariak/igarcia945/InstructDatasets/CrossSum_en.jsonl
47 type: sharegpt
48 conversation: llama3
49 field: conversations
50 roles:
51 input:
52 - system
53 - gpt
54 output:
55 - human
56 - path: /ikerlariak/igarcia945/InstructDatasets/CrossSum_es.jsonl
57 type: sharegpt
58 conversation: llama3
59 field: conversations
60 roles:
61 input:
62 - system
63 - gpt
64 output:
65 - human
66 - path: /ikerlariak/igarcia945/InstructDatasets/Document-Translation-en-es.jsonl
67 type: sharegpt
68 conversation: llama3
69 field: conversations
70 roles:
71 input:
72 - system
73 - gpt
74 output:
75 - human
76 - path: /ikerlariak/igarcia945/InstructDatasets/es-inclusive-language.jsonl
77 type: sharegpt
78 conversation: llama3
79 field: conversations
80 roles:
81 input:
82 - system
83 - gpt
84 output:
85 - human
86 - path: /ikerlariak/igarcia945/InstructDatasets/glaive-code-assistant-v3-small.jsonl
87 type: sharegpt
88 conversation: llama3
89 field: conversations
90 roles:
91 input:
92 - system
93 - gpt
94 output:
95 - human
96 - path: /ikerlariak/igarcia945/InstructDatasets/glaive-function-calling-v2.jsonl
97 type: sharegpt
98 conversation: llama3
99 field: conversations
100 roles:
101 input:
102 - system
103 - gpt
104 - tool
105 output:
106 - human
107 - path: /ikerlariak/igarcia945/InstructDatasets/InstructTranslation-EN-ES.jsonl
108 type: sharegpt
109 conversation: llama3
110 field: conversations
111 roles:
112 input:
113 - system
114 - gpt
115 output:
116 - human
117 - path: /ikerlariak/igarcia945/InstructDatasets/lenguaje-claro-dataset.jsonl
118 type: sharegpt
119 conversation: llama3
120 field: conversations
121 roles:
122 input:
123 - system
124 - gpt
125 output:
126 - human
127 - path: /ikerlariak/igarcia945/InstructDatasets/LingComp_QA.jsonl
128 type: sharegpt
129 conversation: llama3
130 field: conversations
131 roles:
132 input:
133 - system
134 - gpt
135 output:
136 - human
137 - path: /ikerlariak/igarcia945/InstructDatasets/NoticIA.jsonl
138 type: sharegpt
139 conversation: llama3
140 field: conversations
141 roles:
142 input:
143 - system
144 - gpt
145 output:
146 - human
147 - path: /ikerlariak/igarcia945/InstructDatasets/NoticIA-large.jsonl
148 type: sharegpt
149 conversation: llama3
150 field: conversations
151 roles:
152 input:
153 - system
154 - gpt
155 output:
156 - human
157 - path: /ikerlariak/igarcia945/InstructDatasets/NoticIA-summary.jsonl
158 type: sharegpt
159 conversation: llama3
160 field: conversations
161 roles:
162 input:
163 - system
164 - gpt
165 output:
166 - human
167 - path: /ikerlariak/igarcia945/InstructDatasets/OpenHermes-2.5-English.jsonl
168 type: sharegpt
169 conversation: llama3
170 field: conversations
171 roles:
172 input:
173 - system
174 - gpt
175 output:
176 - human
177 - path: /ikerlariak/igarcia945/InstructDatasets/OpenHermes-2.5-Spanish.jsonl
178 type: sharegpt
179 conversation: llama3
180 field: conversations
181 roles:
182 input:
183 - system
184 - gpt
185 output:
186 - human
187 - path: /ikerlariak/igarcia945/InstructDatasets/opus-100-en-es.jsonl
188 type: sharegpt
189 conversation: llama3
190 field: conversations
191 roles:
192 input:
193 - system
194 - gpt
195 output:
196 - human
197 - path: /ikerlariak/igarcia945/InstructDatasets/RAG_Multilingual-es.jsonl
198 type: sharegpt
199 conversation: llama3
200 field: conversations
201 roles:
202 input:
203 - system
204 - gpt
205 output:
206 - human
207 - path: /ikerlariak/igarcia945/InstructDatasets/This-is-not-a-dataset.jsonl
208 type: sharegpt
209 conversation: llama3
210 field: conversations
211 roles:
212 input:
213 - system
214 - gpt
215 output:
216 - human
217 - path: /ikerlariak/igarcia945/InstructDatasets/wikipedia-es.jsonl
218 type: sharegpt
219 conversation: llama3
220 field: conversations
221 roles:
222 input:
223 - system
224 - gpt
225 output:
226 - human
227 - path: /ikerlariak/igarcia945/InstructDatasets/Reddit-Post-Translation.jsonl
228 type: sharegpt
229 conversation: llama3
230 field: conversations
231 roles:
232 input:
233 - system
234 - gpt
235 output:
236 - human
237 - path: /ikerlariak/igarcia945/InstructDatasets/watermark.jsonl
238 type: sharegpt
239 conversation: llama3
240 field: conversations
241 roles:
242 input:
243 - system
244 - gpt
245 output:
246 - human
247
248chat_template: llama3
249
250dataset_prepared_path: /ikerlariak/igarcia945/Mortadelo-Filemon/Meta-Llama-3-8B-Instruct-Spanish-v2/dataset
251
252shuffle_merged_datasets: true
253
254val_set_size: 0.005
255
256output_dir: /ikerlariak/igarcia945/Mortadelo-Filemon/Meta-Llama-3-8B-Instruct-Spanish-v2
257
258adapter:
259lora_model_dir:
260
261sequence_len: 8192
262sample_packing: true
263eval_sample_packing: false
264pad_to_sequence_len: false
265
266tokens:
267 - "<tool_call>"
268 - "<tool_response>"
269 - "<tools>"
270 - "</tool_call>"
271 - "</tool_response>"
272 - "</tools>"
273 - "<reserved1>"
274 - "<reserved2>"
275
276special_tokens:
277 pad_token: <|end_of_text|>
278
279neftune_noise_alpha: 5
280
281wandb_project: Mortadelo&Filemon
282wandb_entity: igarciaf
283wandb_watch:
284wandb_name: Meta-Llama-3-8B-Instruct-Spanish-v2
285wandb_log_model:
286
287gradient_accumulation_steps: 32
288micro_batch_size: 2
289eval_batch_size: 2
290num_epochs: 2
291optimizer: adamw_torch_fused
292lr_scheduler: cosine
293learning_rate: 0.00007
294
295
296train_on_inputs: false
297group_by_length: false
298bf16: true
299fp16: false
300tf32: false
301
302gradient_checkpointing: true
303early_stopping_patience:
304resume_from_checkpoint:
305local_rank:
306logging_steps: 1
307xformers_attention:
308flash_attention: true
309
310warmup_ratio: 0.03
311evals_per_epoch: 4
312eval_table_size:
313save_strategy: "no"
314debug:
315deepspeed: /ikerlariak/igarcia945/Mortadelo-Filemon/train_configs/deepspeed_zero3.json
316weight_decay: 0.0
317fsdp:
318fsdp_config:
319
320seed: 33