Views
No views yet

1# pip install vllm
2# Gemma by default only uses 4k context. You need to set the following variables:
3# export VLLM_WORKER_MULTIPROC_METHOD=spawn
4# export VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
5
6from vllm import LLM, SamplingParams
7
8sampling_params = SamplingParams(
9 best_of=1,
10 temperature=0,
11 max_tokens=8192,
12)
13llm = LLM(model="Unbabel/Tower-Plus-9B", tensor_parallel_size=1)
14messages = [{"role": "user", "content": "Translate the following English source text to Portuguese (Portugal):\nEnglish: Hello world!\nPortuguese (Portugal): "}]
15outputs = llm.chat(messages, sampling_params)
16# Make sure your prompt_token_ids look like this
17print (outputs[0].outputs[0].text)
18# > Olá, mundo!1# Install transformers from source - only needed for versions <= v4.34
2# pip install git+https://github.com/huggingface/transformers.git
3# pip install accelerate
4import torch
5from transformers import pipeline
6
7pipe = pipeline("text-generation", model="Unbabel/Tower-Plus-9B", device_map="auto")
8# We use the tokenizer’s chat template to format each message - see https://huggingface.co/docs/transformers/main/en/chat_templating
9messages = [{"role": "user", "content": "Translate the following English source text to Portuguese (Portugal):\nEnglish: Hello world!\nPortuguese (Portugal): "}]
10input_ids = pipe.tokenizer.apply_chat_template(messages, tokenize=True, add_generation_prompt=True)
11outputs = pipe(messages, max_new_tokens=256, do_sample=False)
12print(outputs[0]["generated_text"])@misc{rei2025towerplus,
title={Tower+: Bridging Generality and Translation Specialization in Multilingual LLMs},
author={Ricardo Rei and Nuno M. Guerreiro and José Pombal and João Alves and Pedro Teixeirinha and Amin Farajian and André F. T. Martins},
year={2025},
eprint={2506.17080},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2506.17080},
}