Views
No views yet


| Model Name | Vision Part | Language Part | HF Link |
|---|---|---|---|
| InternVL3-1B | InternViT-300M-448px-V2_5 | Qwen2.5-0.5B | 🤗 link |
| InternVL3-2B | InternViT-300M-448px-V2_5 | Qwen2.5-1.5B | 🤗 link |
| InternVL3-8B | InternViT-300M-448px-V2_5 | Qwen2.5-7B | 🤗 link |
| InternVL3-9B | InternViT-300M-448px-V2_5 | internlm3-8b-instruct | 🤗 link |
| InternVL3-14B | InternViT-300M-448px-V2_5 | Qwen2.5-14B | 🤗 link |
| InternVL3-38B | InternViT-6B-448px-V2_5 | Qwen2.5-32B | 🤗 link |
| InternVL3-78B | InternViT-6B-448px-V2_5 | Qwen2.5-72B | 🤗 link |















InternVL3-9B using transformers.Please use transformers>=4.37.2 to ensure the model works normally.
1import torch
2from transformers import AutoTokenizer, AutoModel
3path = "OpenGVLab/InternVL3-9B"
4model = AutoModel.from_pretrained(
5 path,
6 torch_dtype=torch.bfloat16,
7 low_cpu_mem_usage=True,
8 use_flash_attn=True,
9 trust_remote_code=True).eval().cuda()1import torch
2from transformers import AutoTokenizer, AutoModel
3path = "OpenGVLab/InternVL3-9B"
4model = AutoModel.from_pretrained(
5 path,
6 torch_dtype=torch.bfloat16,
7 load_in_8bit=True,
8 low_cpu_mem_usage=True,
9 use_flash_attn=True,
10 trust_remote_code=True).eval()1import math
2import torch
3from transformers import AutoTokenizer, AutoModel
4
5def split_model(model_name):
6 device_map = {}
7 world_size = torch.cuda.device_count()
8 config = AutoConfig.from_pretrained(model_path, trust_remote_code=True)
9 num_layers = config.llm_config.num_hidden_layers
10 # Since the first GPU will be used for ViT, treat it as half a GPU.
11 num_layers_per_gpu = math.ceil(num_layers / (world_size - 0.5))
12 num_layers_per_gpu = [num_layers_per_gpu] * world_size
13 num_layers_per_gpu[0] = math.ceil(num_layers_per_gpu[0] * 0.5)
14 layer_cnt = 0
15 for i, num_layer in enumerate(num_layers_per_gpu):
16 for j in range(num_layer):
17 device_map[f'language_model.model.layers.{layer_cnt}'] = i
18 layer_cnt += 1
19 device_map['vision_model'] = 0
20 device_map['mlp1'] = 0
21 device_map['language_model.model.tok_embeddings'] = 0
22 device_map['language_model.model.embed_tokens'] = 0
23 device_map['language_model.output'] = 0
24 device_map['language_model.model.norm'] = 0
25 device_map['language_model.model.rotary_emb'] = 0
26 device_map['language_model.lm_head'] = 0
27 device_map[f'language_model.model.layers.{num_layers - 1}'] = 0
28
29 return device_map
30
31path = "OpenGVLab/InternVL3-9B"
32device_map = split_model('InternVL3-9B')
33model = AutoModel.from_pretrained(
34 path,
35 torch_dtype=torch.bfloat16,
36 low_cpu_mem_usage=True,
37 use_flash_attn=True,
38 trust_remote_code=True,
39 device_map=device_map).eval()1import math
2import numpy as np
3import torch
4import torchvision.transforms as T
5from decord import VideoReader, cpu
6from PIL import Image
7from torchvision.transforms.functional import InterpolationMode
8from transformers import AutoModel, AutoTokenizer
9
10IMAGENET_MEAN = (0.485, 0.456, 0.406)
11IMAGENET_STD = (0.229, 0.224, 0.225)
12
13def build_transform(input_size):
14 MEAN, STD = IMAGENET_MEAN, IMAGENET_STD
15 transform = T.Compose([
16 T.Lambda(lambda img: img.convert('RGB') if img.mode != 'RGB' else img),
17 T.Resize((input_size, input_size), interpolation=InterpolationMode.BICUBIC),
18 T.ToTensor(),
19 T.Normalize(mean=MEAN, std=STD)
20 ])
21 return transform
22
23def find_closest_aspect_ratio(aspect_ratio, target_ratios, width, height, image_size):
24 best_ratio_diff = float('inf')
25 best_ratio = (1, 1)
26 area = width * height
27 for ratio in target_ratios:
28 target_aspect_ratio = ratio[0] / ratio[1]
29 ratio_diff = abs(aspect_ratio - target_aspect_ratio)
30 if ratio_diff < best_ratio_diff:
31 best_ratio_diff = ratio_diff
32 best_ratio = ratio
33 elif ratio_diff == best_ratio_diff:
34 if area > 0.5 * image_size * image_size * ratio[0] * ratio[1]:
35 best_ratio = ratio
36 return best_ratio
37
38def dynamic_preprocess(image, min_num=1, max_num=12, image_size=448, use_thumbnail=False):
39 orig_width, orig_height = image.size
40 aspect_ratio = orig_width / orig_height
41
42 # calculate the existing image aspect ratio
43 target_ratios = set(
44 (i, j) for n in range(min_num, max_num + 1) for i in range(1, n + 1) for j in range(1, n + 1) if
45 i * j <= max_num and i * j >= min_num)
46 target_ratios = sorted(target_ratios, key=lambda x: x[0] * x[1])
47
48 # find the closest aspect ratio to the target
49 target_aspect_ratio = find_closest_aspect_ratio(
50 aspect_ratio, target_ratios, orig_width, orig_height, image_size)
51
52 # calculate the target width and height
53 target_width = image_size * target_aspect_ratio[0]
54 target_height = image_size * target_aspect_ratio[1]
55 blocks = target_aspect_ratio[0] * target_aspect_ratio[1]
56
57 # resize the image
58 resized_img = image.resize((target_width, target_height))
59 processed_images = []
60 for i in range(blocks):
61 box = (
62 (i % (target_width // image_size)) * image_size,
63 (i // (target_width // image_size)) * image_size,
64 ((i % (target_width // image_size)) + 1) * image_size,
65 ((i // (target_width // image_size)) + 1) * image_size
66 )
67 # split the image
68 split_img = resized_img.crop(box)
69 processed_images.append(split_img)
70 assert len(processed_images) == blocks
71 if use_thumbnail and len(processed_images) != 1:
72 thumbnail_img = image.resize((image_size, image_size))
73 processed_images.append(thumbnail_img)
74 return processed_images
75
76def load_image(image_file, input_size=448, max_num=12):
77 image = Image.open(image_file).convert('RGB')
78 transform = build_transform(input_size=input_size)
79 images = dynamic_preprocess(image, image_size=input_size, use_thumbnail=True, max_num=max_num)
80 pixel_values = [transform(image) for image in images]
81 pixel_values = torch.stack(pixel_values)
82 return pixel_values
83
84def split_model(model_name):
85 device_map = {}
86 world_size = torch.cuda.device_count()
87 config = AutoConfig.from_pretrained(model_path, trust_remote_code=True)
88 num_layers = config.llm_config.num_hidden_layers
89 # Since the first GPU will be used for ViT, treat it as half a GPU.
90 num_layers_per_gpu = math.ceil(num_layers / (world_size - 0.5))
91 num_layers_per_gpu = [num_layers_per_gpu] * world_size
92 num_layers_per_gpu[0] = math.ceil(num_layers_per_gpu[0] * 0.5)
93 layer_cnt = 0
94 for i, num_layer in enumerate(num_layers_per_gpu):
95 for j in range(num_layer):
96 device_map[f'language_model.model.layers.{layer_cnt}'] = i
97 layer_cnt += 1
98 device_map['vision_model'] = 0
99 device_map['mlp1'] = 0
100 device_map['language_model.model.tok_embeddings'] = 0
101 device_map['language_model.model.embed_tokens'] = 0
102 device_map['language_model.output'] = 0
103 device_map['language_model.model.norm'] = 0
104 device_map['language_model.model.rotary_emb'] = 0
105 device_map['language_model.lm_head'] = 0
106 device_map[f'language_model.model.layers.{num_layers - 1}'] = 0
107
108 return device_map
109
110# If you set `load_in_8bit=True`, you will need two 80GB GPUs.
111# If you set `load_in_8bit=False`, you will need at least three 80GB GPUs.
112path = 'OpenGVLab/InternVL3-9B'
113device_map = split_model('InternVL3-9B')
114model = AutoModel.from_pretrained(
115 path,
116 torch_dtype=torch.bfloat16,
117 load_in_8bit=False,
118 low_cpu_mem_usage=True,
119 use_flash_attn=True,
120 trust_remote_code=True,
121 device_map=device_map).eval()
122tokenizer = AutoTokenizer.from_pretrained(path, trust_remote_code=True, use_fast=False)
123
124# set the max number of tiles in `max_num`
125pixel_values = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
126generation_config = dict(max_new_tokens=1024, do_sample=True)
127
128# pure-text conversation (纯文本对话)
129question = 'Hello, who are you?'
130response, history = model.chat(tokenizer, None, question, generation_config, history=None, return_history=True)
131print(f'User: {question}\nAssistant: {response}')
132
133question = 'Can you tell me a story?'
134response, history = model.chat(tokenizer, None, question, generation_config, history=history, return_history=True)
135print(f'User: {question}\nAssistant: {response}')
136
137# single-image single-round conversation (单图单轮对话)
138question = '<image>\nPlease describe the image shortly.'
139response = model.chat(tokenizer, pixel_values, question, generation_config)
140print(f'User: {question}\nAssistant: {response}')
141
142# single-image multi-round conversation (单图多轮对话)
143question = '<image>\nPlease describe the image in detail.'
144response, history = model.chat(tokenizer, pixel_values, question, generation_config, history=None, return_history=True)
145print(f'User: {question}\nAssistant: {response}')
146
147question = 'Please write a poem according to the image.'
148response, history = model.chat(tokenizer, pixel_values, question, generation_config, history=history, return_history=True)
149print(f'User: {question}\nAssistant: {response}')
150
151# multi-image multi-round conversation, combined images (多图多轮对话,拼接图像)
152pixel_values1 = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
153pixel_values2 = load_image('./examples/image2.jpg', max_num=12).to(torch.bfloat16).cuda()
154pixel_values = torch.cat((pixel_values1, pixel_values2), dim=0)
155
156question = '<image>\nDescribe the two images in detail.'
157response, history = model.chat(tokenizer, pixel_values, question, generation_config,
158 history=None, return_history=True)
159print(f'User: {question}\nAssistant: {response}')
160
161question = 'What are the similarities and differences between these two images.'
162response, history = model.chat(tokenizer, pixel_values, question, generation_config,
163 history=history, return_history=True)
164print(f'User: {question}\nAssistant: {response}')
165
166# multi-image multi-round conversation, separate images (多图多轮对话,独立图像)
167pixel_values1 = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
168pixel_values2 = load_image('./examples/image2.jpg', max_num=12).to(torch.bfloat16).cuda()
169pixel_values = torch.cat((pixel_values1, pixel_values2), dim=0)
170num_patches_list = [pixel_values1.size(0), pixel_values2.size(0)]
171
172question = 'Image-1: <image>\nImage-2: <image>\nDescribe the two images in detail.'
173response, history = model.chat(tokenizer, pixel_values, question, generation_config,
174 num_patches_list=num_patches_list,
175 history=None, return_history=True)
176print(f'User: {question}\nAssistant: {response}')
177
178question = 'What are the similarities and differences between these two images.'
179response, history = model.chat(tokenizer, pixel_values, question, generation_config,
180 num_patches_list=num_patches_list,
181 history=history, return_history=True)
182print(f'User: {question}\nAssistant: {response}')
183
184# batch inference, single image per sample (单图批处理)
185pixel_values1 = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
186pixel_values2 = load_image('./examples/image2.jpg', max_num=12).to(torch.bfloat16).cuda()
187num_patches_list = [pixel_values1.size(0), pixel_values2.size(0)]
188pixel_values = torch.cat((pixel_values1, pixel_values2), dim=0)
189
190questions = ['<image>\nDescribe the image in detail.'] * len(num_patches_list)
191responses = model.batch_chat(tokenizer, pixel_values,
192 num_patches_list=num_patches_list,
193 questions=questions,
194 generation_config=generation_config)
195for question, response in zip(questions, responses):
196 print(f'User: {question}\nAssistant: {response}')
197
198# video multi-round conversation (视频多轮对话)
199def get_index(bound, fps, max_frame, first_idx=0, num_segments=32):
200 if bound:
201 start, end = bound[0], bound[1]
202 else:
203 start, end = -100000, 100000
204 start_idx = max(first_idx, round(start * fps))
205 end_idx = min(round(end * fps), max_frame)
206 seg_size = float(end_idx - start_idx) / num_segments
207 frame_indices = np.array([
208 int(start_idx + (seg_size / 2) + np.round(seg_size * idx))
209 for idx in range(num_segments)
210 ])
211 return frame_indices
212
213def load_video(video_path, bound=None, input_size=448, max_num=1, num_segments=32):
214 vr = VideoReader(video_path, ctx=cpu(0), num_threads=1)
215 max_frame = len(vr) - 1
216 fps = float(vr.get_avg_fps())
217
218 pixel_values_list, num_patches_list = [], []
219 transform = build_transform(input_size=input_size)
220 frame_indices = get_index(bound, fps, max_frame, first_idx=0, num_segments=num_segments)
221 for frame_index in frame_indices:
222 img = Image.fromarray(vr[frame_index].asnumpy()).convert('RGB')
223 img = dynamic_preprocess(img, image_size=input_size, use_thumbnail=True, max_num=max_num)
224 pixel_values = [transform(tile) for tile in img]
225 pixel_values = torch.stack(pixel_values)
226 num_patches_list.append(pixel_values.shape[0])
227 pixel_values_list.append(pixel_values)
228 pixel_values = torch.cat(pixel_values_list)
229 return pixel_values, num_patches_list
230
231video_path = './examples/red-panda.mp4'
232pixel_values, num_patches_list = load_video(video_path, num_segments=8, max_num=1)
233pixel_values = pixel_values.to(torch.bfloat16).cuda()
234video_prefix = ''.join([f'Frame{i+1}: <image>\n' for i in range(len(num_patches_list))])
235question = video_prefix + 'What is the red panda doing?'
236# Frame1: <image>\nFrame2: <image>\n...\nFrame8: <image>\n{question}
237response, history = model.chat(tokenizer, pixel_values, question, generation_config,
238 num_patches_list=num_patches_list, history=None, return_history=True)
239print(f'User: {question}\nAssistant: {response}')
240
241question = 'Describe this video in detail.'
242response, history = model.chat(tokenizer, pixel_values, question, generation_config,
243 num_patches_list=num_patches_list, history=history, return_history=True)
244print(f'User: {question}\nAssistant: {response}')1from transformers import TextIteratorStreamer
2from threading import Thread
3
4# Initialize the streamer
5streamer = TextIteratorStreamer(tokenizer, skip_prompt=True, skip_special_tokens=True, timeout=10)
6# Define the generation configuration
7generation_config = dict(max_new_tokens=1024, do_sample=False, streamer=streamer)
8# Start the model chat in a separate thread
9thread = Thread(target=model.chat, kwargs=dict(
10 tokenizer=tokenizer, pixel_values=pixel_values, question=question,
11 history=None, return_history=False, generation_config=generation_config,
12))
13thread.start()
14
15# Initialize an empty string to store the generated text
16generated_text = ''
17# Loop through the streamer to get the new text as it is generated
18for new_text in streamer:
19 if new_text == model.conv_template.sep:
20 break
21 generated_text += new_text
22 print(new_text, end='', flush=True) # Print each new chunk of generated text on the same line1# if lmdeploy<0.7.3, you need to explicitly set chat_template_config=ChatTemplateConfig(model_name='internvl2_5')
2pip install lmdeploy>=0.7.31from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL3-9B'
5image = load_image('https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/tests/data/tiger.jpeg')
6pipe = pipeline(model, backend_config=TurbomindEngineConfig(session_len=16384, tp=1), chat_template_config=ChatTemplateConfig(model_name='internvl2_5'))
7response = pipe(('describe this image', image))
8print(response.text)ImportError occurs while executing this case, please install the required dependency packages as prompted.1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3from lmdeploy.vl.constants import IMAGE_TOKEN
4
5model = 'OpenGVLab/InternVL3-9B'
6pipe = pipeline(model, backend_config=TurbomindEngineConfig(session_len=16384, tp=1), chat_template_config=ChatTemplateConfig(model_name='internvl2_5'))
7
8image_urls=[
9 'https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg',
10 'https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/det.jpg'
11]
12
13images = [load_image(img_url) for img_url in image_urls]
14# Numbering images improves multi-image conversations
15response = pipe((f'Image-1: {IMAGE_TOKEN}\nImage-2: {IMAGE_TOKEN}\ndescribe these two images', images))
16print(response.text)1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL3-9B'
5pipe = pipeline(model, backend_config=TurbomindEngineConfig(session_len=16384, tp=1), chat_template_config=ChatTemplateConfig(model_name='internvl2_5'))
6
7image_urls=[
8 "https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg",
9 "https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/det.jpg"
10]
11prompts = [('describe this image', load_image(img_url)) for img_url in image_urls]
12response = pipe(prompts)
13print(response)pipeline.chat interface.1from lmdeploy import pipeline, TurbomindEngineConfig, GenerationConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL3-9B'
5pipe = pipeline(model, backend_config=TurbomindEngineConfig(session_len=16384, tp=1), chat_template_config=ChatTemplateConfig(model_name='internvl2_5'))
6
7image = load_image('https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg')
8gen_config = GenerationConfig(top_k=40, top_p=0.8, temperature=0.8)
9sess = pipe.chat(('describe this image', image), gen_config=gen_config)
10print(sess.response.text)
11sess = pipe.chat('What is the woman doing?', session=sess, gen_config=gen_config)
12print(sess.response.text)api_server enables models to be easily packed into services with a single command. The provided RESTful APIs are compatible with OpenAI's interfaces. Below are an example of service startup:lmdeploy serve api_server OpenGVLab/InternVL3-9B --chat-template internvl2_5 --server-port 23333 --tp 1pip install openai1from openai import OpenAI
2
3client = OpenAI(api_key='YOUR_API_KEY', base_url='http://0.0.0.0:23333/v1')
4model_name = client.models.list().data[0].id
5response = client.chat.completions.create(
6 model=model_name,
7 messages=[{
8 'role':
9 'user',
10 'content': [{
11 'type': 'text',
12 'text': 'describe this image',
13 }, {
14 'type': 'image_url',
15 'image_url': {
16 'url':
17 'https://modelscope.oss-cn-beijing.aliyuncs.com/resource/tiger.jpeg',
18 },
19 }],
20 }],
21 temperature=0.8,
22 top_p=0.8)
23print(response)1@article{chen2024expanding,
2 title={Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling},
3 author={Chen, Zhe and Wang, Weiyun and Cao, Yue and Liu, Yangzhou and Gao, Zhangwei and Cui, Erfei and Zhu, Jinguo and Ye, Shenglong and Tian, Hao and Liu, Zhaoyang and others},
4 journal={arXiv preprint arXiv:2412.05271},
5 year={2024}
6}
7@article{wang2024mpo,
8 title={Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization},
9 author={Wang, Weiyun and Chen, Zhe and Wang, Wenhai and Cao, Yue and Liu, Yangzhou and Gao, Zhangwei and Zhu, Jinguo and Zhu, Xizhou and Lu, Lewei and Qiao, Yu and Dai, Jifeng},
10 journal={arXiv preprint arXiv:2411.10442},
11 year={2024}
12}
13@article{chen2024far,
14 title={How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites},
15 author={Chen, Zhe and Wang, Weiyun and Tian, Hao and Ye, Shenglong and Gao, Zhangwei and Cui, Erfei and Tong, Wenwen and Hu, Kongzhi and Luo, Jiapeng and Ma, Zheng and others},
16 journal={arXiv preprint arXiv:2404.16821},
17 year={2024}
18}
19@inproceedings{chen2024internvl,
20 title={Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks},
21 author={Chen, Zhe and Wu, Jiannan and Wang, Wenhai and Su, Weijie and Chen, Guo and Xing, Sen and Zhong, Muyan and Zhang, Qinglong and Zhu, Xizhou and Lu, Lewei and others},
22 booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition},
23 pages={24185--24198},
24 year={2024}
25}