Views
No views yet

| Dataset | Language | Instances |
|---|---|---|
| TÜLU-v3 | EN | 940,000 |
| LIMA | IT/EN | 2,000 |
| WildChat-IT | IT | 5,000 |
| TowerBlocks-v0.2 | IT/EN | 7,276 |
| GPT-4o-ITA-Instruct | IT | 15,000 |
| Aya | IT | 700 |
| Model | MMLU (5-shots) | ARC-C (5-shots) | Hellaswag (0-shots) | IFEval (inst_level) |
|---|---|---|---|---|
| Llama-3.1-SAVA | 56.9 | 42.3 | 58.1 | 62.3 |
| Llama-3.1-LAPT | 58.5 | 47.9 | 62.4 | 67.3 |
| Mistral-0.1-SAVA | 51.5 | 41.6 | 57.5 | 61.7 |
| Mistral-0.1-LAPT | 52.9 | 39.9 | 58.4 | 60.0 |
| Llama-3.1-Original | 47.4 | 43.1 | 57.9 | 66.8 |
| Mistral-0.1-Original | 41.6 | 38.9 | 50.0 | 42.2 |
pip install --upgrade transformers.1import transformers
2import torch
3
4model_id = "SemanticAlignment/Mistral-v0.1-Italian-LAPT-instruct"
5
6tokenizer = AutoTokenizer.from_pretrained(model_name)
7
8generator = pipeline(
9 "text-generation",
10 model=model_name,
11 device_map="auto",
12 dtype=torch.bfloat16
13)
14
15conversations.append([
16 {"role": "system", "content": "Sei un assistente utile, rispondi in modo conciso e coerente."},
17 {"role": "user", "content": "Cosa si può fare in una bella giornata di sole?"},
18])
19
20chat_samples = tokenizer.apply_chat_template(conversations, tokenize=False)
21
22# get number of prompt tokens
23prompt_tokens_number = len(tokenizer(chat_samples)["input_ids"])
24
25outputs = generator(
26 conversations,
27 max_new_tokens=2048,
28 eos_token_id=[
29 tokenizer.eos_token_id,
30 tokenizer.convert_tokens_to_ids("<|eot_id|>"),
31 ],
32)
331@misc{moroni2025optimizingllmsitalianreducing,
2 title={Optimizing LLMs for Italian: Reducing Token Fertility and Enhancing Efficiency Through Vocabulary Adaptation},
3 author={Luca Moroni and Giovanni Puccetti and Pere-Lluis Huguet Cabot and Andrei Stefan Bejgu and Edoardo Barba and Alessio Miaschi and Felice Dell'Orletta and Andrea Esuli and Roberto Navigli},
4 year={2025},
5 eprint={2504.17025},
6 archivePrefix={arXiv},
7 primaryClass={cs.CL},
8 url={https://arxiv.org/abs/2504.17025},
9}