Views
No views yet
exclude_layers="lm_head model.visual.* mtp.* *mlp.gate *shared_expert* *shared_expert_gate* *linear_attn.* *self_attn.*"
export CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7
export MODEL_DIR=Qwen/Qwen3.5-397B-A17B
export output_dir=amd/Qwen3.5-397B-A17B-NVFP4
python3 quantize_quark.py \
--model_dir $MODEL_DIR \
--quant_scheme nvfp4\
--num_calib_data 128 \
--multi_gpu balanced \
--exclude_layers $exclude_layers \
--model_export hf_format \
--output_dir $output_dir| Benchmark | Qwen/Qwen3.5-397B-A17B-FP8 | amd/Qwen3.5-397B-A17B-NVFP4(this model) | Recovery |
| gsm8k (flexible-extract) | 95.38 | 94.47 | 99.04% |
lm-evaluation-harness framework, based on the Docker image rocm/vllm-dev:nightly_main_20260603.(Version: 0.4.12) in container first.pip install lm-eval[api]export VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
export CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7
lm_eval --model vllm
--model_args pretrained=amd/Qwen3.5-397B-A17B-NVFP4,tensor_parallel_size=8,max_model_len=262144,gpu_memory_utilization=0.90,max_gen_toks=2048,trust_remote_code=True,reasoning_parser=qwen3 \
--tasks gsm8k \
--num_fewshot 5 \
--batch_size auto