Views
No views yet
| Model | MMLU-Med | MMLU-Pro-Med | MMedBench | MedBullets | MedMCQA | MedQA | MedXpertQA-Text | PubMedQA | SuperGPQA-Medical | Avg. |
|---|---|---|---|---|---|---|---|---|---|---|
| Qwen3-VL-4B | 74.3 | 50.7 | 60.5 | 46.4 | 56.0 | 60.5 | 12.6 | 75.6 | 29.6 | 51.8 |
| Qwen3-VL-8B | 79.8 | 57.4 | 65.9 | 51.3 | 61.1 | 65.9 | 12.8 | 76.2 | 30.2 | 55.6 |
| Lingshu-7B | 75.8 | 53.5 | 64.5 | 57.8 | 56.6 | 64.4 | 16.9 | 76.8 | 29.9 | 55.1 |
| HealthGPT-14B | 80.2 | 63.4 | 63.2 | 39.8 | 63.4 | 66.2 | 11.3 | 68.0 | 25.7 | 53.5 |
| HuatuoGPT-V-34B | 74.7 | 51.8 | 60.7 | 42.7 | 54.7 | 58.8 | 11.4 | 54.7 | 26.5 | 48.4 |
| Hulu-Med-4B | 78.6 | 58.6 | 66.7 | 59.4 | 64.8 | 71.9 | 16.8 | 77.6 | 29.5 | 58.2 |
| Hulu-Med-7B | 79.5 | 60.6 | 72.8 | 61.5 | 67.6 | 73.5 | 19.6 | 77.4 | 31.1 | 60.4 |
| HealthGPT-Pro-4B | 80.4 | 58.4 | 71.6 | 58.0 | 64.4 | 71.5 | 16.2 | 78.4 | 31.4 | 58.9 |
| HealthGPT-Pro-8B | 83.1 | 64.1 | 71.4 | 60.6 | 68.5 | 71.3 | 18.3 | 79.2 | 35.4 | 61.3 |
| Lingshu-32B | 85.5 | 70.4 | 80.8 | 70.2 | 65.6 | 74.3 | 22.6 | 79.2 | 43.8 | 65.8 |
| Hulu-Med-32B | 88.1 | 73.0 | 83.6 | 71.0 | 72.3 | 80.4 | 25.0 | 80.6 | 45.4 | 68.8 |
| HealthGPT-Pro-27B | 93.2 | 73.7 | 85.0 | 80.7 | 74.9 | 84.8 | 35.0 | 80.9 | 47.2 | 72.8 |
| Model | MMMU-Med | VQA-RAD | SLAKE | PathVQA | MedXpertQA-Multimodal | MedFrameQA | OmniMedVQA-Mini | PMC-VQA | M3D-MCQ | CT-RATE-MCQ | AMOS-MM-MCQ | Avg. |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| Qwen3-VL-4B | 44.3 | 59.9 | 77.0 | 53.0 | 13.4 | 40.6 | 74.7 | 53.0 | 57.2 | 58.8 | 49.2 | 52.8 |
| Qwen3-VL-8B | 46.5 | 63.4 | 80.2 | 58.3 | 18.7 | 46.4 | 73.0 | 55.6 | 59.5 | 61.6 | 51.2 | 55.9 |
| Lingshu-7B | 47.3 | 66.7 | 81.9 | 61.0 | 25.5 | 52.6 | 82.4 | 57.2 | 64.1 | 68.3 | 62.7 | 60.9 |
| HealthGPT-14B | 45.5 | 62.6 | 64.2 | 56.0 | 24.1 | 45.3 | 70.2 | 56.4 | 55.2 | 57.3 | 46.5 | 53.0 |
| HuatuoGPT-V-34B | 50.1 | 60.3 | 68.3 | 47.7 | 21.5 | 49.6 | 69.7 | 56.6 | 50.1 | 54.9 | 48.7 | 52.5 |
| Hulu-Med-4B | 45.8 | 72.6 | 81.7 | 59.7 | 24.6 | 54.2 | 75.1 | 53.1 | 76.0 | 70.1 | 69.1 | 62.0 |
| Hulu-Med-7B | 50.5 | 77.2 | 85.8 | 64.2 | 28.3 | 57.4 | 77.7 | 57.3 | 80.4 | 76.2 | 70.5 | 66.0 |
| HealthGPT-Pro-4B | 52.0 | 76.6 | 83.9 | 66.7 | 20.8 | 61.4 | 78.2 | 60.0 | 81.0 | 86.2 | 71.1 | 67.1 |
| HealthGPT-Pro-8B | 54.7 | 78.4 | 85.0 | 70.7 | 25.3 | 63.6 | 80.2 | 61.1 | 81.6 | 86.0 | 72.2 | 69.0 |
| Lingshu-32B | 62.1 | 68.9 | 89.9 | 85.8 | 30.4 | 60.3 | 83.5 | 65.2 | 64.8 | 74.3 | 65.9 | 68.3 |
| Hulu-Med-32B | 42.8 | 72.9 | 91.6 | 90.0 | 35.7 | 60.9 | 81.2 | 64.3 | 79.3 | 82.8 | 75.0 | 70.6 |
| HealthGPT-Pro-27B | 68.3 | 68.9 | 92.6 | 87.2 | 44.1 | 62.6 | 79.8 | 70.0 | 84.9 | 89.9 | 80.3 | 75.3 |
1# Create and activate a clean Python 3.12 environment
2conda create -n healthgpt-pro python=3.12 -y
3conda activate healthgpt-pro
4
5# Install PyTorch with CUDA support
6# If your CUDA version is lower than 12.8, install a matching PyTorch build instead (e.g., cu121 or cu118).
7pip install torch==2.8.0 torchvision==0.23.0 torchaudio==2.8.0 --index-url https://download.pytorch.org/whl/cu128
8
9# Install FlashAttention for faster attention
10pip install flash-attn==2.8.3 --no-build-isolation --upgrade
11
12# Install other dependencies
13pip install transformers==4.57.1 accelerate==1.11.0 deepspeed==0.16.9 numpy==1.26.4 peft==0.17.1
14pip install qwen-vl-utils pillow1import numpy as np
2import torch
3from PIL import Image
4from transformers import AutoProcessor, Qwen3VLForConditionalGeneration
5from qwen_vl_utils import process_vision_info
6
7model_id = "HealthGPT-Pro-8B"
8
9model = Qwen3VLForConditionalGeneration.from_pretrained(
10 model_id,
11 dtype=torch.bfloat16,
12 attn_implementation="flash_attention_2",
13 device_map="auto",
14)
15processor = AutoProcessor.from_pretrained(model_id)1messages = [
2 {
3 "role": "user",
4 "content": [
5 {"type": "text", "text": "Explain the key symptoms and common risk factors of pneumonia."},
6 ],
7 }
8]
9
10inputs = processor.apply_chat_template(
11 messages,
12 tokenize=True,
13 add_generation_prompt=True,
14 return_dict=True,
15 return_tensors="pt",
16).to(model.device)
17
18generated_ids = model.generate(**inputs, max_new_tokens=256)
19generated_ids_trimmed = [
20 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
21]
22output_text = processor.batch_decode(
23 generated_ids_trimmed,
24 skip_special_tokens=True,
25 clean_up_tokenization_spaces=False,
26)
27print(output_text[0])1messages = [
2 {
3 "role": "user",
4 "content": [
5 {"type": "image", "image": "examples/chest_xray.png"},
6 {"type": "text", "text": "Describe the main radiological findings in this image."},
7 ],
8 }
9]
10
11inputs = processor.apply_chat_template(
12 messages,
13 tokenize=True,
14 add_generation_prompt=True,
15 return_dict=True,
16 return_tensors="pt",
17).to(model.device)
18
19generated_ids = model.generate(**inputs, max_new_tokens=256)
20generated_ids_trimmed = [
21 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
22]
23output_text = processor.batch_decode(
24 generated_ids_trimmed,
25 skip_special_tokens=True,
26 clean_up_tokenization_spaces=False,
27)
28print(output_text[0])1messages = [
2 {
3 "role": "user",
4 "content": [
5 {"type": "image", "image": "examples/image_1.png"},
6 {"type": "image", "image": "examples/image_2.png"},
7 {"type": "text", "text": "Compare these two medical images and summarize the key differences."},
8 ],
9 }
10]
11
12inputs = processor.apply_chat_template(
13 messages,
14 tokenize=True,
15 add_generation_prompt=True,
16 return_dict=True,
17 return_tensors="pt",
18).to(model.device)
19
20generated_ids = model.generate(**inputs, max_new_tokens=256)
21generated_ids_trimmed = [
22 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
23]
24output_text = processor.batch_decode(
25 generated_ids_trimmed,
26 skip_special_tokens=True,
27 clean_up_tokenization_spaces=False,
28)
29print(output_text[0]).npy volume into a sequence of 2D frames and sends it as a video-style input.1def ct_to_video(ct_path: str):
2 ct_pixels = np.load(ct_path)
3 ct_u8 = np.clip(ct_pixels * 255, 0, 255).astype(np.uint8)
4
5 frames = []
6 idx = np.linspace(1, len(ct_u8) - 2, 10, dtype=int)
7 for i in idx:
8 rgb = np.stack([ct_u8[i]] * 3, axis=-1)
9 frames.append(Image.fromarray(rgb, mode="RGB"))
10 return frames
11
12volume_frames = ct_to_video("examples/ct_volume.npy")
13messages = [
14 {
15 "role": "user",
16 "content": [
17 {"type": "video", "video": volume_frames, "sample_fps": 2.0},
18 {"type": "text", "text": "Analyze this CT volume and summarize the main findings."},
19 ],
20 }
21]
22
23text = processor.apply_chat_template(
24 messages,
25 tokenize=False,
26 add_generation_prompt=True,
27)
28images, videos, video_kwargs = process_vision_info(
29 messages,
30 image_patch_size=16,
31 return_video_kwargs=True,
32 return_video_metadata=True,
33)
34if videos is not None:
35 videos, video_metadatas = zip(*videos)
36 videos, video_metadatas = list(videos), list(video_metadatas)
37else:
38 video_metadatas = None
39
40inputs = processor(
41 text=text,
42 images=images,
43 videos=videos,
44 video_metadata=video_metadatas,
45 return_tensors="pt",
46 do_resize=False,
47 **video_kwargs,
48).to(model.device)
49
50generated_ids = model.generate(**inputs, max_new_tokens=256)
51generated_ids_trimmed = [
52 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
53]
54output_text = processor.batch_decode(
55 generated_ids_trimmed,
56 skip_special_tokens=True,
57 clean_up_tokenization_spaces=False,
58)
59print(output_text[0])1@misc{lin2025healthgptmedicallargevisionlanguage,
2 title={HealthGPT: A Medical Large Vision-Language Model for Unifying Comprehension and Generation via Heterogeneous Knowledge Adaptation},
3 author={Tianwei Lin and Wenqiao Zhang and Sijing Li and Yuqian Yuan and Binhe Yu and Haoyuan Li and Wanggui He and Hao Jiang and Mengze Li and Xiaohui Song and Siliang Tang and Jun Xiao and Hui Lin and Yueting Zhuang and Beng Chin Ooi},
4 year={2025},
5 eprint={2502.09838},
6 archivePrefix={arXiv},
7 primaryClass={cs.CV},
8 url={https://arxiv.org/abs/2502.09838},
9}