Views
No views yet
vllm serve RedHatAI/Qwen3-VL-235B-A22B-Instruct-FP8-dynamic --tensor_parallel_size 41from openai import OpenAI
2
3# Modify OpenAI's API key and API base to use vLLM's API server.
4openai_api_key = "EMPTY"
5openai_api_base = "http://<your-server-host>:8000/v1"
6
7client = OpenAI(
8 api_key=openai_api_key,
9 base_url=openai_api_base,
10)
11
12model = "RedHatAI/Qwen3-VL-235B-A22B-Instruct-FP8-dynamic"
13
14messages = [
15 {
16 "role": "user",
17 "content": [
18 {
19 "type": "image_url",
20 "image_url": {"url": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg"},
21 },
22 {"type": "text", "text": "Describe this image."},
23 ],
24 }
25]
26
27outputs = client.chat.completions.create(
28 model=model,
29 messages=messages,
30)
31
32generated_text = outputs.choices[0].message.content
33print(generated_text)1from transformers import AutoProcessor, Qwen3VLMoeForConditionalGeneration
2
3from llmcompressor import oneshot
4from llmcompressor.modeling import replace_modules_for_calibration
5from llmcompressor.modifiers.quantization import QuantizationModifier
6
7MODEL_ID = "Qwen/Qwen3-VL-235B-A22B-Instruct"
8
9# Load model.
10model = Qwen3VLMoeForConditionalGeneration.from_pretrained(MODEL_ID, torch_dtype="auto")
11processor = AutoProcessor.from_pretrained(MODEL_ID)
12model = replace_modules_for_calibration(model)
13
14# Configure the quantization algorithm and scheme.
15# In this case, we:
16# * quantize the weights to fp8 with per-channel quantization
17# * quantize the activations to fp8 with dynamic token activations
18recipe = QuantizationModifier(
19 targets="Linear",
20 scheme="FP8_DYNAMIC",
21 ignore=[
22 "re:.*lm_head",
23 "re:visual.*",
24 "re:model.visual.*",
25 "re:.*mlp.gate$",
26 ],
27)
28
29# Apply quantization.
30oneshot(model=model, recipe=recipe)
31
32# Save to disk in compressed-tensors format.
33SAVE_DIR = MODEL_ID.rstrip("/").split("/")[-1] + "-FP8-dynamic"
34model.save_pretrained(SAVE_DIR)
35processor.save_pretrained(SAVE_DIR)lm_eval \
--model vllm \
--model_args pretrained="RedHatAI/Qwen3-VL-235B-A22B-Instruct-FP8-dynamic",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=4,gpu_memory_utilization=0.8,enable_chunked_prefill=True \
--tasks openllm \
--write_out \
--batch_size auto \
--output_path output_dir \
--show_config1model_parameters:
2 model_name: RedHatAI/Qwen3-VL-235B-A22B-Instruct-FP8-dynamic
3 dtype: auto
4 gpu_memory_utilization: 0.9
5 generation_parameters:
6 temperature: 0.6
7 min_p: 0.0
8 top_p: 0.95
9 top_k: 20
10 max_new_tokens: 32768lighteval vllm \
--model_args lighteval_model_arguments.yaml \
--tasks lighteval|aime25|0 \python3 -m lmms_eval \
--model vllm \
--model_args model=RedHatAI/Qwen3-VL-235B-A22B-Instruct-FP8-dynamic,tensor_parallel_size=4,max_model_len=8192,gpu_memory_utilization=0.9 \
--tasks mmmu_val, chartqa\
--batch_size 1| Category | Metric | Qwen3-VL-235B-A22B-Instruct | Qwen3-VL-235B-A22B-Instruct-FP8-dynamic | Recovery (%) |
|---|---|---|---|---|
| OpenLLM V1 | ARC-Challenge (Acc-Norm, 25-shot) | 76.54 | 75.94 | 99.2 |
| GSM8K (Strict-Match, 5-shot) | 90.30 | 89.92 | 99.6 | |
| HellaSwag (Acc-Norm, 10-shot) | 87.81 | 87.74 | 99.9 | |
| MMLU (Acc, 5-shot) | 87.11 | 87.23 | 100.1 | |
| TruthfulQA (MC2, 0-shot) | 63.19 | 63.48 | 100.5 | |
| Winogrande (Acc, 5-shot) | 81.61 | 82.64 | 101.3 | |
| Average Score | 81.09 | 81.16 | 100.1 | |
| Reasoning (generation) | AIME 2025 | 70.00 | 80.00 | 114.3 |
| Multi-modal | ChartQA (relaxed_overall) | 90.12 | 90.04 | 99.9 |
| MMMU (val) | 63.67 | 63.33 | 99.5 |