Views
No views yet

| Model | MMMU | DocVQA_VAL | MMBench_DEV_EN | MathVista_MINI |
|---|---|---|---|---|
| Qwen2.5-VL-32B-Instruct | 70.0 | 93.9107 | 87.3 | 74.7 |
| Qwen2.5-VL-32B-Instruct-AWQ | 67.8 | 94.1489 | 86.9 | 73.6 |
pip install git+https://github.com/huggingface/transformers accelerate
KeyError: 'qwen2_5_vl'
pip install git+https://github.com/huggingface/transformers accelerate
KeyError: 'qwen2_5_vl'
1# It's highly recommanded to use `[decord]` feature for faster video loading.
2pip install qwen-vl-utils[decord]==0.0.8decord from PyPI. In that case, you can use pip install qwen-vl-utils which will fall back to using torchvision for video processing. However, you can still install decord from source to get decord used when loading video.transformers and qwen_vl_utils:1from transformers import Qwen2_5_VLForConditionalGeneration, AutoTokenizer, AutoProcessor
2from qwen_vl_utils import process_vision_info
3
4# default: Load the model on the available device(s)
5model = Qwen2_5_VLForConditionalGeneration.from_pretrained(
6 "Qwen/Qwen2.5-VL-32B-Instruct-AWQ", torch_dtype="auto", device_map="auto"
7)
8
9# We recommend enabling flash_attention_2 for better acceleration and memory saving, especially in multi-image and video scenarios.
10# model = Qwen2_5_VLForConditionalGeneration.from_pretrained(
11# "Qwen/Qwen2.5-VL-32B-Instruct-AWQ",
12# torch_dtype=torch.bfloat16,
13# attn_implementation="flash_attention_2",
14# device_map="auto",
15# )
16
17# default processer
18processor = AutoProcessor.from_pretrained("Qwen/Qwen2.5-VL-32B-Instruct-AWQ")
19
20# The default range for the number of visual tokens per image in the model is 4-16384.
21# You can set min_pixels and max_pixels according to your needs, such as a token range of 256-1280, to balance performance and cost.
22# min_pixels = 256*28*28
23# max_pixels = 1280*28*28
24# processor = AutoProcessor.from_pretrained("Qwen/Qwen2.5-VL-32B-Instruct-AWQ", min_pixels=min_pixels, max_pixels=max_pixels)
25
26messages = [
27 {
28 "role": "user",
29 "content": [
30 {
31 "type": "image",
32 "image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
33 },
34 {"type": "text", "text": "Describe this image."},
35 ],
36 }
37]
38
39# Preparation for inference
40text = processor.apply_chat_template(
41 messages, tokenize=False, add_generation_prompt=True
42)
43image_inputs, video_inputs = process_vision_info(messages)
44inputs = processor(
45 text=[text],
46 images=image_inputs,
47 videos=video_inputs,
48 padding=True,
49 return_tensors="pt",
50)
51inputs = inputs.to("cuda")
52
53# Inference: Generation of the output
54generated_ids = model.generate(**inputs, max_new_tokens=128)
55generated_ids_trimmed = [
56 out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
57]
58output_text = processor.batch_decode(
59 generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
60)
61print(output_text)1# Messages containing multiple images and a text query
2messages = [
3 {
4 "role": "user",
5 "content": [
6 {"type": "image", "image": "file:///path/to/image1.jpg"},
7 {"type": "image", "image": "file:///path/to/image2.jpg"},
8 {"type": "text", "text": "Identify the similarities between these images."},
9 ],
10 }
11]
12
13# Preparation for inference
14text = processor.apply_chat_template(
15 messages, tokenize=False, add_generation_prompt=True
16)
17image_inputs, video_inputs = process_vision_info(messages)
18inputs = processor(
19 text=[text],
20 images=image_inputs,
21 videos=video_inputs,
22 padding=True,
23 return_tensors="pt",
24)
25inputs = inputs.to("cuda")
26
27# Inference
28generated_ids = model.generate(**inputs, max_new_tokens=128)
29generated_ids_trimmed = [
30 out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
31]
32output_text = processor.batch_decode(
33 generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
34)
35print(output_text)1# Messages containing a images list as a video and a text query
2messages = [
3 {
4 "role": "user",
5 "content": [
6 {
7 "type": "video",
8 "video": [
9 "file:///path/to/frame1.jpg",
10 "file:///path/to/frame2.jpg",
11 "file:///path/to/frame3.jpg",
12 "file:///path/to/frame4.jpg",
13 ],
14 },
15 {"type": "text", "text": "Describe this video."},
16 ],
17 }
18]
19
20# Messages containing a local video path and a text query
21messages = [
22 {
23 "role": "user",
24 "content": [
25 {
26 "type": "video",
27 "video": "file:///path/to/video1.mp4",
28 "max_pixels": 360 * 420,
29 "fps": 1.0,
30 },
31 {"type": "text", "text": "Describe this video."},
32 ],
33 }
34]
35
36# Messages containing a video url and a text query
37messages = [
38 {
39 "role": "user",
40 "content": [
41 {
42 "type": "video",
43 "video": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen2-VL/space_woaudio.mp4",
44 },
45 {"type": "text", "text": "Describe this video."},
46 ],
47 }
48]
49
50#In Qwen 2.5 VL, frame rate information is also input into the model to align with absolute time.
51# Preparation for inference
52text = processor.apply_chat_template(
53 messages, tokenize=False, add_generation_prompt=True
54)
55image_inputs, video_inputs, video_kwargs = process_vision_info(messages, return_video_kwargs=True)
56inputs = processor(
57 text=[text],
58 images=image_inputs,
59 videos=video_inputs,
60 fps=fps,
61 padding=True,
62 return_tensors="pt",
63 **video_kwargs,
64)
65inputs = inputs.to("cuda")
66
67# Inference
68generated_ids = model.generate(**inputs, max_new_tokens=128)
69generated_ids_trimmed = [
70 out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
71]
72output_text = processor.batch_decode(
73 generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
74)
75print(output_text)FORCE_QWENVL_VIDEO_READER=torchvision or FORCE_QWENVL_VIDEO_READER=decord if you prefer not to use the default one.| Backend | HTTP | HTTPS |
|---|---|---|
| torchvision >= 0.19.0 | ✅ | ✅ |
| torchvision < 0.19.0 | ❌ | ❌ |
| decord | ✅ | ❌ |
1# Sample messages for batch inference
2messages1 = [
3 {
4 "role": "user",
5 "content": [
6 {"type": "image", "image": "file:///path/to/image1.jpg"},
7 {"type": "image", "image": "file:///path/to/image2.jpg"},
8 {"type": "text", "text": "What are the common elements in these pictures?"},
9 ],
10 }
11]
12messages2 = [
13 {"role": "system", "content": "You are a helpful assistant."},
14 {"role": "user", "content": "Who are you?"},
15]
16# Combine messages for batch processing
17messages = [messages1, messages2]
18
19# Preparation for batch inference
20texts = [
21 processor.apply_chat_template(msg, tokenize=False, add_generation_prompt=True)
22 for msg in messages
23]
24image_inputs, video_inputs = process_vision_info(messages)
25inputs = processor(
26 text=texts,
27 images=image_inputs,
28 videos=video_inputs,
29 padding=True,
30 return_tensors="pt",
31)
32inputs = inputs.to("cuda")
33
34# Batch Inference
35generated_ids = model.generate(**inputs, max_new_tokens=128)
36generated_ids_trimmed = [
37 out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
38]
39output_texts = processor.batch_decode(
40 generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
41)
42print(output_texts)snapshot_download can help you solve issues concerning downloading checkpoints.1# You can directly insert a local file path, a URL, or a base64-encoded image into the position where you want in the text.
2## Local file path
3messages = [
4 {
5 "role": "user",
6 "content": [
7 {"type": "image", "image": "file:///path/to/your/image.jpg"},
8 {"type": "text", "text": "Describe this image."},
9 ],
10 }
11]
12## Image URL
13messages = [
14 {
15 "role": "user",
16 "content": [
17 {"type": "image", "image": "http://path/to/your/image.jpg"},
18 {"type": "text", "text": "Describe this image."},
19 ],
20 }
21]
22## Base64 encoded image
23messages = [
24 {
25 "role": "user",
26 "content": [
27 {"type": "image", "image": "data:image;base64,/9j/..."},
28 {"type": "text", "text": "Describe this image."},
29 ],
30 }
31]1min_pixels = 256 * 28 * 28
2max_pixels = 1280 * 28 * 28
3processor = AutoProcessor.from_pretrained(
4 "Qwen/Qwen2.5-VL-32B-Instruct-AWQ", min_pixels=min_pixels, max_pixels=max_pixels
5)resized_height and resized_width. These values will be rounded to the nearest multiple of 28.1# min_pixels and max_pixels
2messages = [
3 {
4 "role": "user",
5 "content": [
6 {
7 "type": "image",
8 "image": "file:///path/to/your/image.jpg",
9 "resized_height": 280,
10 "resized_width": 420,
11 },
12 {"type": "text", "text": "Describe this image."},
13 ],
14 }
15]
16# resized_height and resized_width
17messages = [
18 {
19 "role": "user",
20 "content": [
21 {
22 "type": "image",
23 "image": "file:///path/to/your/image.jpg",
24 "min_pixels": 50176,
25 "max_pixels": 50176,
26 },
27 {"type": "text", "text": "Describe this image."},
28 ],
29 }
30]config.json is set for context length up to 32,768 tokens.
To handle extensive inputs exceeding 32,768 tokens, we utilize YaRN, a technique for enhancing model length extrapolation, ensuring optimal performance on lengthy texts.config.json to enable YaRN:@article{Qwen2.5-VL,
title={Qwen2.5-VL Technical Report},
author={Bai, Shuai and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Song, Sibo and Dang, Kai and Wang, Peng and Wang, Shijie and Tang, Jun and Zhong, Humen and Zhu, Yuanzhi and Yang, Mingkun and Li, Zhaohai and Wan, Jianqiang and Wang, Pengfei and Ding, Wei and Fu, Zheren and Xu, Yiheng and Ye, Jiabo and Zhang, Xi and Xie, Tianbao and Cheng, Zesen and Zhang, Hang and Yang, Zhibo and Xu, Haiyang and Lin, Junyang},
journal={arXiv preprint arXiv:2502.13923},
year={2025}
}