Views
No views yet
| Model | MMLU-Med | MMLU-Pro-Med | MMedBench | MedBullets | MedMCQA | MedQA | MedXpertQA-Text | PubMedQA | SuperGPQA-Medical | Avg. |
|---|---|---|---|---|---|---|---|---|---|---|
| Qwen3-VL-4B | 74.3 | 50.7 | 60.5 | 46.4 | 56.0 | 60.5 | 12.6 | 75.6 | 29.6 | 51.8 |
| Qwen3-VL-8B | 79.8 | 57.4 | 65.9 | 51.3 | 61.1 | 65.9 | 12.8 | 76.2 | 30.2 | 55.6 |
| Lingshu-7B | 75.8 | 53.5 | 64.5 | 57.8 | 56.6 | 64.4 | 16.9 | 76.8 | 29.9 | 55.1 |
| HealthGPT-14B | 80.2 | 63.4 | 63.2 | 39.8 | 63.4 | 66.2 | 11.3 | 68.0 | 25.7 | 53.5 |
| HuatuoGPT-V-34B | 74.7 | 51.8 | 60.7 | 42.7 | 54.7 | 58.8 | 11.4 | 54.7 | 26.5 | 48.4 |
| Hulu-Med-4B | 78.6 | 58.6 | 66.7 | 59.4 | 64.8 | 71.9 | 16.8 | 77.6 | 29.5 | 58.2 |
| Hulu-Med-7B | 79.5 | 60.6 | 72.8 | 61.5 | 67.6 | 73.5 | 19.6 | 77.4 | 31.1 | 60.4 |
| HealthGPT-Pro-4B | 80.4 | 58.4 | 71.6 | 58.0 | 64.4 | 71.5 | 16.2 | 78.4 | 31.4 | 58.9 |
| HealthGPT-Pro-8B | 83.1 | 64.1 | 71.4 | 60.6 | 68.5 | 71.3 | 18.3 | 79.2 | 35.4 | 61.3 |
| Lingshu-32B | 85.5 | 70.4 | 80.8 | 70.2 | 65.6 | 74.3 | 22.6 | 79.2 | 43.8 | 65.8 |
| Hulu-Med-32B | 88.1 | 73.0 | 83.6 | 71.0 | 72.3 | 80.4 | 25.0 | 80.6 | 45.4 | 68.8 |
| HealthGPT-Pro-27B | 93.2 | 73.7 | 85.0 | 80.7 | 74.9 | 84.8 | 35.0 | 80.9 | 47.2 | 72.8 |
| Model | MMMU-Med | VQA-RAD | SLAKE | PathVQA | MedXpertQA-Multimodal | MedFrameQA | OmniMedVQA-Mini | PMC-VQA | M3D-MCQ | CT-RATE-MCQ | AMOS-MM-MCQ | Avg. |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| Qwen3-VL-4B | 44.3 | 59.9 | 77.0 | 53.0 | 13.4 | 40.6 | 74.7 | 53.0 | 57.2 | 58.8 | 49.2 | 52.8 |
| Qwen3-VL-8B | 46.5 | 63.4 | 80.2 | 58.3 | 18.7 | 46.4 | 73.0 | 55.6 | 59.5 | 61.6 | 51.2 | 55.9 |
| Lingshu-7B | 47.3 | 66.7 | 81.9 | 61.0 | 25.5 | 52.6 | 82.4 | 57.2 | 64.1 | 68.3 | 62.7 | 60.9 |
| HealthGPT-14B | 45.5 | 62.6 | 64.2 | 56.0 | 24.1 | 45.3 | 70.2 | 56.4 | 55.2 | 57.3 | 46.5 | 53.0 |
| HuatuoGPT-V-34B | 50.1 | 60.3 | 68.3 | 47.7 | 21.5 | 49.6 | 69.7 | 56.6 | 50.1 | 54.9 | 48.7 | 52.5 |
| Hulu-Med-4B | 45.8 | 72.6 | 81.7 | 59.7 | 24.6 | 54.2 | 75.1 | 53.1 | 76.0 | 70.1 | 69.1 | 62.0 |
| Hulu-Med-7B | 50.5 | 77.2 | 85.8 | 64.2 | 28.3 | 57.4 | 77.7 | 57.3 | 80.4 | 76.2 | 70.5 | 66.0 |
| HealthGPT-Pro-4B | 52.0 | 76.6 | 83.9 | 66.7 | 20.8 | 61.4 | 78.2 | 60.0 | 81.0 | 86.2 | 71.1 | 67.1 |
| HealthGPT-Pro-8B | 54.7 | 78.4 | 85.0 | 70.7 | 25.3 | 63.6 | 80.2 | 61.1 | 81.6 | 86.0 | 72.2 | 69.0 |
| Lingshu-32B | 62.1 | 68.9 | 89.9 | 85.8 | 30.4 | 60.3 | 83.5 | 65.2 | 64.8 | 74.3 | 65.9 | 68.3 |
| Hulu-Med-32B | 42.8 | 72.9 | 91.6 | 90.0 | 35.7 | 60.9 | 81.2 | 64.3 | 79.3 | 82.8 | 75.0 | 70.6 |
| HealthGPT-Pro-27B | 68.3 | 68.9 | 92.6 | 87.2 | 44.1 | 62.6 | 79.8 | 70.0 | 84.9 | 89.9 | 80.3 | 75.3 |
1# Create and activate a clean Python 3.11 environment
2conda create -n healthgpt-pro python=3.11 -y
3conda activate healthgpt-pro
4
5# Install PyTorch with CUDA support. Choose the wheel matching your CUDA/runtime.
6pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu128
7
8# Install inference dependencies
9pip install transformers accelerate qwen-vl-utils pillow
10
11# Optional: install FlashAttention for faster attention if your CUDA/PyTorch stack supports it.
12pip install flash-attn --no-build-isolation --upgrade1import numpy as np
2import torch
3from PIL import Image
4from transformers import AutoProcessor, Qwen3_5ForConditionalGeneration
5from qwen_vl_utils import process_vision_info
6
7model_id = "lintw/HealthGPT-Pro-27B"
8
9model = Qwen3_5ForConditionalGeneration.from_pretrained(
10 model_id,
11 dtype=torch.bfloat16,
12 attn_implementation="flash_attention_2",
13 device_map="auto",
14)
15processor = AutoProcessor.from_pretrained(model_id)1messages = [
2 {
3 "role": "user",
4 "content": [
5 {"type": "text", "text": "Explain the key symptoms and common risk factors of pneumonia."},
6 ],
7 }
8]
9
10inputs = processor.apply_chat_template(
11 messages,
12 tokenize=True,
13 add_generation_prompt=True,
14 return_dict=True,
15 return_tensors="pt",
16).to(model.device)
17
18generated_ids = model.generate(**inputs, max_new_tokens=256)
19generated_ids_trimmed = [
20 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
21]
22output_text = processor.batch_decode(
23 generated_ids_trimmed,
24 skip_special_tokens=True,
25 clean_up_tokenization_spaces=False,
26)
27print(output_text[0])1messages = [
2 {
3 "role": "user",
4 "content": [
5 {"type": "image", "image": "examples/chest_xray.png"},
6 {"type": "text", "text": "Describe the main radiological findings in this image."},
7 ],
8 }
9]
10
11inputs = processor.apply_chat_template(
12 messages,
13 tokenize=True,
14 add_generation_prompt=True,
15 return_dict=True,
16 return_tensors="pt",
17).to(model.device)
18
19generated_ids = model.generate(**inputs, max_new_tokens=256)
20generated_ids_trimmed = [
21 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
22]
23output_text = processor.batch_decode(
24 generated_ids_trimmed,
25 skip_special_tokens=True,
26 clean_up_tokenization_spaces=False,
27)
28print(output_text[0])1messages = [
2 {
3 "role": "user",
4 "content": [
5 {"type": "image", "image": "examples/image_1.png"},
6 {"type": "image", "image": "examples/image_2.png"},
7 {"type": "text", "text": "Compare these two medical images and summarize the key differences."},
8 ],
9 }
10]
11
12inputs = processor.apply_chat_template(
13 messages,
14 tokenize=True,
15 add_generation_prompt=True,
16 return_dict=True,
17 return_tensors="pt",
18).to(model.device)
19
20generated_ids = model.generate(**inputs, max_new_tokens=256)
21generated_ids_trimmed = [
22 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
23]
24output_text = processor.batch_decode(
25 generated_ids_trimmed,
26 skip_special_tokens=True,
27 clean_up_tokenization_spaces=False,
28)
29print(output_text[0]).npy volume into a sequence of 2D frames and sends it as a video-style input.1def ct_to_video(ct_path: str):
2 ct_pixels = np.load(ct_path)
3 ct_u8 = np.clip(ct_pixels * 255, 0, 255).astype(np.uint8)
4
5 frames = []
6 idx = np.linspace(1, len(ct_u8) - 2, 10, dtype=int)
7 for i in idx:
8 rgb = np.stack([ct_u8[i]] * 3, axis=-1)
9 frames.append(Image.fromarray(rgb, mode="RGB"))
10 return frames
11
12volume_frames = ct_to_video("examples/ct_volume.npy")
13messages = [
14 {
15 "role": "user",
16 "content": [
17 {"type": "video", "video": volume_frames, "sample_fps": 2.0},
18 {"type": "text", "text": "Analyze this CT volume and summarize the main findings."},
19 ],
20 }
21]
22
23text = processor.apply_chat_template(
24 messages,
25 tokenize=False,
26 add_generation_prompt=True,
27)
28images, videos, video_kwargs = process_vision_info(
29 messages,
30 image_patch_size=16,
31 return_video_kwargs=True,
32 return_video_metadata=True,
33)
34if videos is not None:
35 videos, video_metadatas = zip(*videos)
36 videos, video_metadatas = list(videos), list(video_metadatas)
37else:
38 video_metadatas = None
39
40inputs = processor(
41 text=text,
42 images=images,
43 videos=videos,
44 video_metadata=video_metadatas,
45 return_tensors="pt",
46 do_resize=False,
47 **video_kwargs,
48).to(model.device)
49
50generated_ids = model.generate(**inputs, max_new_tokens=256)
51generated_ids_trimmed = [
52 out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
53]
54output_text = processor.batch_decode(
55 generated_ids_trimmed,
56 skip_special_tokens=True,
57 clean_up_tokenization_spaces=False,
58)
59print(output_text[0])1@misc{lin2025healthgptmedicallargevisionlanguage,
2 title={HealthGPT: A Medical Large Vision-Language Model for Unifying Comprehension and Generation via Heterogeneous Knowledge Adaptation},
3 author={Tianwei Lin and Wenqiao Zhang and Sijing Li and Yuqian Yuan and Binhe Yu and Haoyuan Li and Wanggui He and Hao Jiang and Mengze Li and Xiaohui Song and Siliang Tang and Jun Xiao and Hui Lin and Yueting Zhuang and Beng Chin Ooi},
4 year={2025},
5 eprint={2502.09838},
6 archivePrefix={arXiv},
7 primaryClass={cs.CV},
8 url={https://arxiv.org/abs/2502.09838},
9}