Views
No views yet

| Model Name | Vision Part | Language Part | HF Link | MS Link |
|---|---|---|---|---|
| InternVL2-1B | InternViT-300M-448px | Qwen2-0.5B-Instruct | 🤗 link | 🤖 link |
| InternVL2-2B | InternViT-300M-448px | internlm2-chat-1_8b | 🤗 link | 🤖 link |
| InternVL2-4B | InternViT-300M-448px | Phi-3-mini-128k-instruct | 🤗 link | 🤖 link |
| InternVL2-8B | InternViT-300M-448px | internlm2_5-7b-chat | 🤗 link | 🤖 link |
| InternVL2-26B | InternViT-6B-448px-V1-5 | internlm2-chat-20b | 🤗 link | 🤖 link |
| InternVL2-40B | InternViT-6B-448px-V1-5 | Nous-Hermes-2-Yi-34B | 🤗 link | 🤖 link |
| InternVL2-Llama3-76B | InternViT-6B-448px-V1-5 | Hermes-2-Theta-Llama-3-70B | 🤗 link | 🤖 link |
| Benchmark | MiniCPM-Llama3-V-2_5 | InternVL-Chat-V1-5 | InternVL2-8B |
|---|---|---|---|
| Model Size | 8.5B | 25.5B | 8.1B |
| DocVQAtest | 84.8 | 90.9 | 91.6 |
| ChartQAtest | - | 83.8 | 83.3 |
| InfoVQAtest | - | 72.5 | 74.8 |
| TextVQAval | 76.6 | 80.6 | 77.4 |
| OCRBench | 725 | 724 | 794 |
| MMEsum | 2024.6 | 2187.8 | 2210.3 |
| RealWorldQA | 63.5 | 66.0 | 64.4 |
| AI2Dtest | 78.4 | 80.7 | 83.8 |
| MMMUval | 45.8 | 45.2 / 46.8 | 49.3 / 51.8 |
| MMBench-ENtest | 77.2 | 82.2 | 81.7 |
| MMBench-CNtest | 74.2 | 82.0 | 81.2 |
| CCBenchdev | 45.9 | 69.8 | 75.9 |
| MMVetGPT-4-0613 | - | 62.8 | 60.0 |
| MMVetGPT-4-Turbo | 52.8 | 55.4 | 54.2 |
| SEED-Image | 72.3 | 76.0 | 76.2 |
| HallBenchavg | 42.4 | 49.3 | 45.2 |
| MathVistatestmini | 54.3 | 53.5 | 58.3 |
| OpenCompassavg | 58.8 | 61.7 | 64.1 |
| Benchmark | VideoChat2-HD-Mistral | Video-CCAM-9B | InternVL2-4B | InternVL2-8B |
|---|---|---|---|---|
| Model Size | 7B | 9B | 4.2B | 8.1B |
| MVBench | 60.4 | 60.7 | 63.7 | 66.4 |
| MMBench-Video8f | - | - | 1.10 | 1.19 |
| MMBench-Video16f | - | - | 1.18 | 1.28 |
| Video-MME w/o subs | 42.3 | 50.6 | 51.4 | 54.0 |
| Video-MME w subs | 54.6 | 54.9 | 53.4 | 56.9 |
| Model | avg. | RefCOCO (val) | RefCOCO (testA) | RefCOCO (testB) | RefCOCO+ (val) | RefCOCO+ (testA) | RefCOCO+ (testB) | RefCOCO‑g (val) | RefCOCO‑g (test) |
|---|---|---|---|---|---|---|---|---|---|
| UNINEXT-H (Specialist SOTA) | 88.9 | 92.6 | 94.3 | 91.5 | 85.2 | 89.6 | 79.8 | 88.7 | 89.4 |
| Mini-InternVL- Chat-2B-V1-5 | 75.8 | 80.7 | 86.7 | 72.9 | 72.5 | 82.3 | 60.8 | 75.6 | 74.9 |
| Mini-InternVL- Chat-4B-V1-5 | 84.4 | 88.0 | 91.4 | 83.5 | 81.5 | 87.4 | 73.8 | 84.7 | 84.6 |
| InternVL‑Chat‑V1‑5 | 88.8 | 91.4 | 93.7 | 87.1 | 87.0 | 92.3 | 80.9 | 88.5 | 89.3 |
| InternVL2‑1B | 79.9 | 83.6 | 88.7 | 79.8 | 76.0 | 83.6 | 67.7 | 80.2 | 79.9 |
| InternVL2‑2B | 77.7 | 82.3 | 88.2 | 75.9 | 73.5 | 82.8 | 63.3 | 77.6 | 78.3 |
| InternVL2‑4B | 84.4 | 88.5 | 91.2 | 83.9 | 81.2 | 87.2 | 73.8 | 84.6 | 84.6 |
| InternVL2‑8B | 82.9 | 87.1 | 91.1 | 80.7 | 79.8 | 87.9 | 71.4 | 82.7 | 82.7 |
| InternVL2‑26B | 88.5 | 91.2 | 93.3 | 87.4 | 86.8 | 91.0 | 81.2 | 88.5 | 88.6 |
| InternVL2‑40B | 90.3 | 93.0 | 94.7 | 89.2 | 88.5 | 92.8 | 83.6 | 90.3 | 90.6 |
| InternVL2- Llama3‑76B | 90.0 | 92.2 | 94.8 | 88.4 | 88.8 | 93.1 | 82.8 | 89.5 | 90.3 |
Please provide the bounding box coordinates of the region this sentence describes: <ref>{}</ref>transformers.Please use transformers==4.37.2 to ensure the model works normally.
1import torch
2from transformers import AutoTokenizer, AutoModel
3path = "OpenGVLab/InternVL2-8B"
4model = AutoModel.from_pretrained(
5 path,
6 torch_dtype=torch.bfloat16,
7 low_cpu_mem_usage=True,
8 trust_remote_code=True).eval().cuda()1import torch
2from transformers import AutoTokenizer, AutoModel
3path = "OpenGVLab/InternVL2-8B"
4model = AutoModel.from_pretrained(
5 path,
6 torch_dtype=torch.bfloat16,
7 load_in_8bit=True,
8 low_cpu_mem_usage=True,
9 trust_remote_code=True).eval()1import torch
2from transformers import AutoTokenizer, AutoModel
3path = "OpenGVLab/InternVL2-8B"
4model = AutoModel.from_pretrained(
5 path,
6 torch_dtype=torch.bfloat16,
7 load_in_4bit=True,
8 low_cpu_mem_usage=True,
9 trust_remote_code=True).eval()1import math
2import torch
3from transformers import AutoTokenizer, AutoModel
4
5def split_model(model_name):
6 device_map = {}
7 world_size = torch.cuda.device_count()
8 num_layers = {
9 'InternVL2-1B': 24, 'InternVL2-2B': 24, 'InternVL2-4B': 32, 'InternVL2-8B': 32,
10 'InternVL2-26B': 48, 'InternVL2-40B': 60, 'InternVL2-Llama3-76B': 80}[model_name]
11 # Since the first GPU will be used for ViT, treat it as half a GPU.
12 num_layers_per_gpu = math.ceil(num_layers / (world_size - 0.5))
13 num_layers_per_gpu = [num_layers_per_gpu] * world_size
14 num_layers_per_gpu[0] = math.ceil(num_layers_per_gpu[0] * 0.5)
15 layer_cnt = 0
16 for i, num_layer in enumerate(num_layers_per_gpu):
17 for j in range(num_layer):
18 device_map[f'language_model.model.layers.{layer_cnt}'] = i
19 layer_cnt += 1
20 device_map['vision_model'] = 0
21 device_map['mlp1'] = 0
22 device_map['language_model.model.tok_embeddings'] = 0
23 device_map['language_model.model.embed_tokens'] = 0
24 device_map['language_model.output'] = 0
25 device_map['language_model.model.norm'] = 0
26 device_map['language_model.lm_head'] = 0
27 device_map[f'language_model.model.layers.{num_layers - 1}'] = 0
28
29 return device_map
30
31path = "OpenGVLab/InternVL2-8B"
32device_map = split_model('InternVL2-8B')
33model = AutoModel.from_pretrained(
34 path,
35 torch_dtype=torch.bfloat16,
36 low_cpu_mem_usage=True,
37 trust_remote_code=True,
38 device_map=device_map).eval()1import numpy as np
2import torch
3import torchvision.transforms as T
4from decord import VideoReader, cpu
5from PIL import Image
6from torchvision.transforms.functional import InterpolationMode
7from transformers import AutoModel, AutoTokenizer
8
9IMAGENET_MEAN = (0.485, 0.456, 0.406)
10IMAGENET_STD = (0.229, 0.224, 0.225)
11
12def build_transform(input_size):
13 MEAN, STD = IMAGENET_MEAN, IMAGENET_STD
14 transform = T.Compose([
15 T.Lambda(lambda img: img.convert('RGB') if img.mode != 'RGB' else img),
16 T.Resize((input_size, input_size), interpolation=InterpolationMode.BICUBIC),
17 T.ToTensor(),
18 T.Normalize(mean=MEAN, std=STD)
19 ])
20 return transform
21
22def find_closest_aspect_ratio(aspect_ratio, target_ratios, width, height, image_size):
23 best_ratio_diff = float('inf')
24 best_ratio = (1, 1)
25 area = width * height
26 for ratio in target_ratios:
27 target_aspect_ratio = ratio[0] / ratio[1]
28 ratio_diff = abs(aspect_ratio - target_aspect_ratio)
29 if ratio_diff < best_ratio_diff:
30 best_ratio_diff = ratio_diff
31 best_ratio = ratio
32 elif ratio_diff == best_ratio_diff:
33 if area > 0.5 * image_size * image_size * ratio[0] * ratio[1]:
34 best_ratio = ratio
35 return best_ratio
36
37def dynamic_preprocess(image, min_num=1, max_num=12, image_size=448, use_thumbnail=False):
38 orig_width, orig_height = image.size
39 aspect_ratio = orig_width / orig_height
40
41 # calculate the existing image aspect ratio
42 target_ratios = set(
43 (i, j) for n in range(min_num, max_num + 1) for i in range(1, n + 1) for j in range(1, n + 1) if
44 i * j <= max_num and i * j >= min_num)
45 target_ratios = sorted(target_ratios, key=lambda x: x[0] * x[1])
46
47 # find the closest aspect ratio to the target
48 target_aspect_ratio = find_closest_aspect_ratio(
49 aspect_ratio, target_ratios, orig_width, orig_height, image_size)
50
51 # calculate the target width and height
52 target_width = image_size * target_aspect_ratio[0]
53 target_height = image_size * target_aspect_ratio[1]
54 blocks = target_aspect_ratio[0] * target_aspect_ratio[1]
55
56 # resize the image
57 resized_img = image.resize((target_width, target_height))
58 processed_images = []
59 for i in range(blocks):
60 box = (
61 (i % (target_width // image_size)) * image_size,
62 (i // (target_width // image_size)) * image_size,
63 ((i % (target_width // image_size)) + 1) * image_size,
64 ((i // (target_width // image_size)) + 1) * image_size
65 )
66 # split the image
67 split_img = resized_img.crop(box)
68 processed_images.append(split_img)
69 assert len(processed_images) == blocks
70 if use_thumbnail and len(processed_images) != 1:
71 thumbnail_img = image.resize((image_size, image_size))
72 processed_images.append(thumbnail_img)
73 return processed_images
74
75def load_image(image_file, input_size=448, max_num=12):
76 image = Image.open(image_file).convert('RGB')
77 transform = build_transform(input_size=input_size)
78 images = dynamic_preprocess(image, image_size=input_size, use_thumbnail=True, max_num=max_num)
79 pixel_values = [transform(image) for image in images]
80 pixel_values = torch.stack(pixel_values)
81 return pixel_values
82
83# If you want to load a model using multiple GPUs, please refer to the `Multiple GPUs` section.
84path = 'OpenGVLab/InternVL2-8B'
85model = AutoModel.from_pretrained(
86 path,
87 torch_dtype=torch.bfloat16,
88 low_cpu_mem_usage=True,
89 trust_remote_code=True).eval().cuda()
90tokenizer = AutoTokenizer.from_pretrained(path, trust_remote_code=True, use_fast=False)
91
92# set the max number of tiles in `max_num`
93pixel_values = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
94generation_config = dict(max_new_tokens=1024, do_sample=False)
95
96# pure-text conversation (纯文本对话)
97question = 'Hello, who are you?'
98response, history = model.chat(tokenizer, None, question, generation_config, history=None, return_history=True)
99print(f'User: {question}\nAssistant: {response}')
100
101question = 'Can you tell me a story?'
102response, history = model.chat(tokenizer, None, question, generation_config, history=history, return_history=True)
103print(f'User: {question}\nAssistant: {response}')
104
105# single-image single-round conversation (单图单轮对话)
106question = '<image>\nPlease describe the image shortly.'
107response = model.chat(tokenizer, pixel_values, question, generation_config)
108print(f'User: {question}\nAssistant: {response}')
109
110# single-image multi-round conversation (单图多轮对话)
111question = '<image>\nPlease describe the image in detail.'
112response, history = model.chat(tokenizer, pixel_values, question, generation_config, history=None, return_history=True)
113print(f'User: {question}\nAssistant: {response}')
114
115question = 'Please write a poem according to the image.'
116response, history = model.chat(tokenizer, pixel_values, question, generation_config, history=history, return_history=True)
117print(f'User: {question}\nAssistant: {response}')
118
119# multi-image multi-round conversation, combined images (多图多轮对话,拼接图像)
120pixel_values1 = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
121pixel_values2 = load_image('./examples/image2.jpg', max_num=12).to(torch.bfloat16).cuda()
122pixel_values = torch.cat((pixel_values1, pixel_values2), dim=0)
123
124question = '<image>\nDescribe the two images in detail.'
125response, history = model.chat(tokenizer, pixel_values, question, generation_config,
126 history=None, return_history=True)
127print(f'User: {question}\nAssistant: {response}')
128
129question = 'What are the similarities and differences between these two images.'
130response, history = model.chat(tokenizer, pixel_values, question, generation_config,
131 history=history, return_history=True)
132print(f'User: {question}\nAssistant: {response}')
133
134# multi-image multi-round conversation, separate images (多图多轮对话,独立图像)
135pixel_values1 = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
136pixel_values2 = load_image('./examples/image2.jpg', max_num=12).to(torch.bfloat16).cuda()
137pixel_values = torch.cat((pixel_values1, pixel_values2), dim=0)
138num_patches_list = [pixel_values1.size(0), pixel_values2.size(0)]
139
140question = 'Image-1: <image>\nImage-2: <image>\nDescribe the two images in detail.'
141response, history = model.chat(tokenizer, pixel_values, question, generation_config,
142 num_patches_list=num_patches_list,
143 history=None, return_history=True)
144print(f'User: {question}\nAssistant: {response}')
145
146question = 'What are the similarities and differences between these two images.'
147response, history = model.chat(tokenizer, pixel_values, question, generation_config,
148 num_patches_list=num_patches_list,
149 history=history, return_history=True)
150print(f'User: {question}\nAssistant: {response}')
151
152# batch inference, single image per sample (单图批处理)
153pixel_values1 = load_image('./examples/image1.jpg', max_num=12).to(torch.bfloat16).cuda()
154pixel_values2 = load_image('./examples/image2.jpg', max_num=12).to(torch.bfloat16).cuda()
155num_patches_list = [pixel_values1.size(0), pixel_values2.size(0)]
156pixel_values = torch.cat((pixel_values1, pixel_values2), dim=0)
157
158questions = ['<image>\nDescribe the image in detail.'] * len(num_patches_list)
159responses = model.batch_chat(tokenizer, pixel_values,
160 num_patches_list=num_patches_list,
161 questions=questions,
162 generation_config=generation_config)
163for question, response in zip(questions, responses):
164 print(f'User: {question}\nAssistant: {response}')
165
166# video multi-round conversation (视频多轮对话)
167def get_index(bound, fps, max_frame, first_idx=0, num_segments=32):
168 if bound:
169 start, end = bound[0], bound[1]
170 else:
171 start, end = -100000, 100000
172 start_idx = max(first_idx, round(start * fps))
173 end_idx = min(round(end * fps), max_frame)
174 seg_size = float(end_idx - start_idx) / num_segments
175 frame_indices = np.array([
176 int(start_idx + (seg_size / 2) + np.round(seg_size * idx))
177 for idx in range(num_segments)
178 ])
179 return frame_indices
180
181def load_video(video_path, bound=None, input_size=448, max_num=1, num_segments=32):
182 vr = VideoReader(video_path, ctx=cpu(0), num_threads=1)
183 max_frame = len(vr) - 1
184 fps = float(vr.get_avg_fps())
185
186 pixel_values_list, num_patches_list = [], []
187 transform = build_transform(input_size=input_size)
188 frame_indices = get_index(bound, fps, max_frame, first_idx=0, num_segments=num_segments)
189 for frame_index in frame_indices:
190 img = Image.fromarray(vr[frame_index].asnumpy()).convert('RGB')
191 img = dynamic_preprocess(img, image_size=input_size, use_thumbnail=True, max_num=max_num)
192 pixel_values = [transform(tile) for tile in img]
193 pixel_values = torch.stack(pixel_values)
194 num_patches_list.append(pixel_values.shape[0])
195 pixel_values_list.append(pixel_values)
196 pixel_values = torch.cat(pixel_values_list)
197 return pixel_values, num_patches_list
198
199video_path = './examples/red-panda.mp4'
200pixel_values, num_patches_list = load_video(video_path, num_segments=8, max_num=1)
201pixel_values = pixel_values.to(torch.bfloat16).cuda()
202video_prefix = ''.join([f'Frame{i+1}: <image>\n' for i in range(len(num_patches_list))])
203question = video_prefix + 'What is the red panda doing?'
204# Frame1: <image>\nFrame2: <image>\n...\nFrame8: <image>\n{question}
205response, history = model.chat(tokenizer, pixel_values, question, generation_config,
206 num_patches_list=num_patches_list, history=None, return_history=True)
207print(f'User: {question}\nAssistant: {response}')
208
209question = 'Describe this video in detail. Don\'t repeat.'
210response, history = model.chat(tokenizer, pixel_values, question, generation_config,
211 num_patches_list=num_patches_list, history=history, return_history=True)
212print(f'User: {question}\nAssistant: {response}')1from transformers import TextIteratorStreamer
2from threading import Thread
3
4# Initialize the streamer
5streamer = TextIteratorStreamer(tokenizer, skip_prompt=True, skip_special_tokens=True, timeout=10)
6# Define the generation configuration
7generation_config = dict(max_new_tokens=1024, do_sample=False, streamer=streamer)
8# Start the model chat in a separate thread
9thread = Thread(target=model.chat, kwargs=dict(
10 tokenizer=tokenizer, pixel_values=pixel_values, question=question,
11 history=None, return_history=False, generation_config=generation_config,
12))
13thread.start()
14
15# Initialize an empty string to store the generated text
16generated_text = ''
17# Loop through the streamer to get the new text as it is generated
18for new_text in streamer:
19 if new_text == model.conv_template.sep:
20 break
21 generated_text += new_text
22 print(new_text, end='', flush=True) # Print each new chunk of generated text on the same linepip install lmdeploy1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL2-8B'
5system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
6image = load_image('https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/tests/data/tiger.jpeg')
7chat_template_config = ChatTemplateConfig('internvl-internlm2')
8chat_template_config.meta_instruction = system_prompt
9pipe = pipeline(model, chat_template_config=chat_template_config,
10 backend_config=TurbomindEngineConfig(session_len=8192))
11response = pipe(('describe this image', image))
12print(response.text)ImportError occurs while executing this case, please install the required dependency packages as prompted.Warning: Due to the scarcity of multi-image conversation data, the performance on multi-image tasks may be unstable, and it may require multiple attempts to achieve satisfactory results.
1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3from lmdeploy.vl.constants import IMAGE_TOKEN
4
5model = 'OpenGVLab/InternVL2-8B'
6system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
7chat_template_config = ChatTemplateConfig('internvl-internlm2')
8chat_template_config.meta_instruction = system_prompt
9pipe = pipeline(model, chat_template_config=chat_template_config,
10 backend_config=TurbomindEngineConfig(session_len=8192))
11
12image_urls=[
13 'https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg',
14 'https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/det.jpg'
15]
16
17images = [load_image(img_url) for img_url in image_urls]
18# Numbering images improves multi-image conversations
19response = pipe((f'Image-1: {IMAGE_TOKEN}\nImage-2: {IMAGE_TOKEN}\ndescribe these two images', images))
20print(response.text)1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL2-8B'
5system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
6chat_template_config = ChatTemplateConfig('internvl-internlm2')
7chat_template_config.meta_instruction = system_prompt
8pipe = pipeline(model, chat_template_config=chat_template_config,
9 backend_config=TurbomindEngineConfig(session_len=8192))
10
11image_urls=[
12 "https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg",
13 "https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/det.jpg"
14]
15prompts = [('describe this image', load_image(img_url)) for img_url in image_urls]
16response = pipe(prompts)
17print(response)pipeline.chat interface.1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig, GenerationConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL2-8B'
5system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
6chat_template_config = ChatTemplateConfig('internvl-internlm2')
7chat_template_config.meta_instruction = system_prompt
8pipe = pipeline(model, chat_template_config=chat_template_config,
9 backend_config=TurbomindEngineConfig(session_len=8192))
10
11image = load_image('https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg')
12gen_config = GenerationConfig(top_k=40, top_p=0.8, temperature=0.8)
13sess = pipe.chat(('describe this image', image), gen_config=gen_config)
14print(sess.response.text)
15sess = pipe.chat('What is the woman doing?', session=sess, gen_config=gen_config)
16print(sess.response.text)chat_template.json.1{
2 "model_name":"internvl-internlm2",
3 "meta_instruction":"我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。",
4 "stop_words":["<|im_start|>", "<|im_end|>"]
5}api_server enables models to be easily packed into services with a single command. The provided RESTful APIs are compatible with OpenAI's interfaces. Below are an example of service startup:lmdeploy serve api_server OpenGVLab/InternVL2-8B --backend turbomind --server-port 23333 --chat-template chat_template.jsonpip install openai1from openai import OpenAI
2
3client = OpenAI(api_key='YOUR_API_KEY', base_url='http://0.0.0.0:23333/v1')
4model_name = client.models.list().data[0].id
5response = client.chat.completions.create(
6 model=model_name,
7 messages=[{
8 'role':
9 'user',
10 'content': [{
11 'type': 'text',
12 'text': 'describe this image',
13 }, {
14 'type': 'image_url',
15 'image_url': {
16 'url':
17 'https://modelscope.oss-cn-beijing.aliyuncs.com/resource/tiger.jpeg',
18 },
19 }],
20 }],
21 temperature=0.8,
22 top_p=0.8)
23print(response)1@article{chen2023internvl,
2 title={InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks},
3 author={Chen, Zhe and Wu, Jiannan and Wang, Wenhai and Su, Weijie and Chen, Guo and Xing, Sen and Zhong, Muyan and Zhang, Qinglong and Zhu, Xizhou and Lu, Lewei and Li, Bin and Luo, Ping and Lu, Tong and Qiao, Yu and Dai, Jifeng},
4 journal={arXiv preprint arXiv:2312.14238},
5 year={2023}
6}
7@article{chen2024far,
8 title={How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites},
9 author={Chen, Zhe and Wang, Weiyun and Tian, Hao and Ye, Shenglong and Gao, Zhangwei and Cui, Erfei and Tong, Wenwen and Hu, Kongzhi and Luo, Jiapeng and Ma, Zheng and others},
10 journal={arXiv preprint arXiv:2404.16821},
11 year={2024}
12}| 模型名称 | 视觉部分 | 语言部分 | HF 链接 | MS 链接 |
|---|---|---|---|---|
| InternVL2-1B | InternViT-300M-448px | Qwen2-0.5B-Instruct | 🤗 link | 🤖 link |
| InternVL2-2B | InternViT-300M-448px | internlm2-chat-1_8b | 🤗 link | 🤖 link |
| InternVL2-4B | InternViT-300M-448px | Phi-3-mini-128k-instruct | 🤗 link | 🤖 link |
| InternVL2-8B | InternViT-300M-448px | internlm2_5-7b-chat | 🤗 link | 🤖 link |
| InternVL2-26B | InternViT-6B-448px-V1-5 | internlm2-chat-20b | 🤗 link | 🤖 link |
| InternVL2-40B | InternViT-6B-448px-V1-5 | Nous-Hermes-2-Yi-34B | 🤗 link | 🤖 link |
| InternVL2-Llama3-76B | InternViT-6B-448px-V1-5 | Hermes-2-Theta-Llama-3-70B | 🤗 link | 🤖 link |
| 评测数据集 | MiniCPM-Llama3-V-2_5 | InternVL-Chat-V1-5 | InternVL2-8B |
|---|---|---|---|
| 模型大小 | 8.5B | 25.5B | 8.1B |
| DocVQAtest | 84.8 | 90.9 | 91.6 |
| ChartQAtest | - | 83.8 | 83.3 |
| InfoVQAtest | - | 72.5 | 74.8 |
| TextVQAval | 76.6 | 80.6 | 77.4 |
| OCRBench | 725 | 724 | 794 |
| MMEsum | 2024.6 | 2187.8 | 2210.3 |
| RealWorldQA | 63.5 | 66.0 | 64.4 |
| AI2Dtest | 78.4 | 80.7 | 83.8 |
| MMMUval | 45.8 | 45.2 / 46.8 | 49.3 / 51.8 |
| MMBench-ENtest | 77.2 | 82.2 | 81.7 |
| MMBench-CNtest | 74.2 | 82.0 | 81.2 |
| CCBenchdev | 45.9 | 69.8 | 75.9 |
| MMVetGPT-4-0613 | - | 62.8 | 60.0 |
| MMVetGPT-4-Turbo | 52.8 | 55.4 | 54.2 |
| SEED-Image | 72.3 | 76.0 | 76.2 |
| HallBenchavg | 42.4 | 49.3 | 45.2 |
| MathVistatestmini | 54.3 | 53.5 | 58.3 |
| OpenCompassavg | 58.8 | 61.7 | 64.1 |
| 评测数据集 | VideoChat2-HD-Mistral | Video-CCAM-9B | InternVL2-4B | InternVL2-8B |
|---|---|---|---|---|
| 模型大小 | 7B | 9B | 4.2B | 8.1B |
| MVBench | 60.4 | 60.7 | 63.7 | 66.4 |
| MMBench-Video8f | - | - | 1.10 | 1.19 |
| MMBench-Video16f | - | - | 1.18 | 1.28 |
| Video-MME w/o subs | 42.3 | 50.6 | 51.4 | 54.0 |
| Video-MME w subs | 54.6 | 54.9 | 53.4 | 56.9 |
| 模型 | avg. | RefCOCO (val) | RefCOCO (testA) | RefCOCO (testB) | RefCOCO+ (val) | RefCOCO+ (testA) | RefCOCO+ (testB) | RefCOCO‑g (val) | RefCOCO‑g (test) |
|---|---|---|---|---|---|---|---|---|---|
| UNINEXT-H (Specialist SOTA) | 88.9 | 92.6 | 94.3 | 91.5 | 85.2 | 89.6 | 79.8 | 88.7 | 89.4 |
| Mini-InternVL- Chat-2B-V1-5 | 75.8 | 80.7 | 86.7 | 72.9 | 72.5 | 82.3 | 60.8 | 75.6 | 74.9 |
| Mini-InternVL- Chat-4B-V1-5 | 84.4 | 88.0 | 91.4 | 83.5 | 81.5 | 87.4 | 73.8 | 84.7 | 84.6 |
| InternVL‑Chat‑V1‑5 | 88.8 | 91.4 | 93.7 | 87.1 | 87.0 | 92.3 | 80.9 | 88.5 | 89.3 |
| InternVL2‑1B | 79.9 | 83.6 | 88.7 | 79.8 | 76.0 | 83.6 | 67.7 | 80.2 | 79.9 |
| InternVL2‑2B | 77.7 | 82.3 | 88.2 | 75.9 | 73.5 | 82.8 | 63.3 | 77.6 | 78.3 |
| InternVL2‑4B | 84.4 | 88.5 | 91.2 | 83.9 | 81.2 | 87.2 | 73.8 | 84.6 | 84.6 |
| InternVL2‑8B | 82.9 | 87.1 | 91.1 | 80.7 | 79.8 | 87.9 | 71.4 | 82.7 | 82.7 |
| InternVL2‑26B | 88.5 | 91.2 | 93.3 | 87.4 | 86.8 | 91.0 | 81.2 | 88.5 | 88.6 |
| InternVL2‑40B | 90.3 | 93.0 | 94.7 | 89.2 | 88.5 | 92.8 | 83.6 | 90.3 | 90.6 |
| InternVL2- Llama3‑76B | 90.0 | 92.2 | 94.8 | 88.4 | 88.8 | 93.1 | 82.8 | 89.5 | 90.3 |
Please provide the bounding box coordinates of the region this sentence describes: <ref>{}</ref>transformers 运行 InternVL2-8B。请使用 transformers==4.37.2 以确保模型正常运行。
pip install lmdeploy1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL2-8B'
5system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
6image = load_image('https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/tests/data/tiger.jpeg')
7chat_template_config = ChatTemplateConfig('internvl-internlm2')
8chat_template_config.meta_instruction = system_prompt
9pipe = pipeline(model, chat_template_config=chat_template_config,
10 backend_config=TurbomindEngineConfig(session_len=8192))
11response = pipe(('describe this image', image))
12print(response.text)ImportError,请按照提示安装所需的依赖包。1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3from lmdeploy.vl.constants import IMAGE_TOKEN
4
5model = 'OpenGVLab/InternVL2-8B'
6system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
7chat_template_config = ChatTemplateConfig('internvl-internlm2')
8chat_template_config.meta_instruction = system_prompt
9pipe = pipeline(model, chat_template_config=chat_template_config,
10 backend_config=TurbomindEngineConfig(session_len=8192))
11
12image_urls=[
13 'https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg',
14 'https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/det.jpg'
15]
16
17images = [load_image(img_url) for img_url in image_urls]
18response = pipe((f'Image-1: {IMAGE_TOKEN}\nImage-2: {IMAGE_TOKEN}\ndescribe these two images', images))
19print(response.text)1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL2-8B'
5system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
6chat_template_config = ChatTemplateConfig('internvl-internlm2')
7chat_template_config.meta_instruction = system_prompt
8pipe = pipeline(model, chat_template_config=chat_template_config,
9 backend_config=TurbomindEngineConfig(session_len=8192))
10
11image_urls=[
12 "https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg",
13 "https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/det.jpg"
14]
15prompts = [('describe this image', load_image(img_url)) for img_url in image_urls]
16response = pipe(prompts)
17print(response)pipeline.chat 接口。1from lmdeploy import pipeline, TurbomindEngineConfig, ChatTemplateConfig, GenerationConfig
2from lmdeploy.vl import load_image
3
4model = 'OpenGVLab/InternVL2-8B'
5system_prompt = '我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。'
6chat_template_config = ChatTemplateConfig('internvl-internlm2')
7chat_template_config.meta_instruction = system_prompt
8pipe = pipeline(model, chat_template_config=chat_template_config,
9 backend_config=TurbomindEngineConfig(session_len=8192))
10
11image = load_image('https://raw.githubusercontent.com/open-mmlab/mmdeploy/main/demo/resources/human-pose.jpg')
12gen_config = GenerationConfig(top_k=40, top_p=0.8, temperature=0.8)
13sess = pipe.chat(('describe this image', image), gen_config=gen_config)
14print(sess.response.text)
15sess = pipe.chat('What is the woman doing?', session=sess, gen_config=gen_config)
16print(sess.response.text)chat_template.json。1{
2 "model_name":"internvl-internlm2",
3 "meta_instruction":"我是书生·万象,英文名是InternVL,是由上海人工智能实验室、清华大学及多家合作单位联合开发的多模态大语言模型。",
4 "stop_words":["<|im_start|>", "<|im_end|>"]
5}api_server 使模型能够通过一个命令轻松打包成服务。提供的 RESTful API 与 OpenAI 的接口兼容。以下是服务启动的示例:lmdeploy serve api_server OpenGVLab/InternVL2-8B --backend turbomind --server-port 23333 --chat-template chat_template.jsonpip install openai1from openai import OpenAI
2
3client = OpenAI(api_key='YOUR_API_KEY', base_url='http://0.0.0.0:23333/v1')
4model_name = client.models.list().data[0].id
5response = client.chat.completions.create(
6 model=model_name,
7 messages=[{
8 'role':
9 'user',
10 'content': [{
11 'type': 'text',
12 'text': 'describe this image',
13 }, {
14 'type': 'image_url',
15 'image_url': {
16 'url':
17 'https://modelscope.oss-cn-beijing.aliyuncs.com/resource/tiger.jpeg',
18 },
19 }],
20 }],
21 temperature=0.8,
22 top_p=0.8)
23print(response)1@article{chen2023internvl,
2 title={InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks},
3 author={Chen, Zhe and Wu, Jiannan and Wang, Wenhai and Su, Weijie and Chen, Guo and Xing, Sen and Zhong, Muyan and Zhang, Qinglong and Zhu, Xizhou and Lu, Lewei and Li, Bin and Luo, Ping and Lu, Tong and Qiao, Yu and Dai, Jifeng},
4 journal={arXiv preprint arXiv:2312.14238},
5 year={2023}
6}
7@article{chen2024far,
8 title={How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites},
9 author={Chen, Zhe and Wang, Weiyun and Tian, Hao and Ye, Shenglong and Gao, Zhangwei and Cui, Erfei and Tong, Wenwen and Hu, Kongzhi and Luo, Jiapeng and Ma, Zheng and others},
10 journal={arXiv preprint arXiv:2404.16821},
11 year={2024}
12}