Views
No views yet
| Attribute | Value |
|---|---|
| Base Model | Qwen/Qwen3.5-35B-A3B |
| Quantization Tool | AutoRound |
| Quantization Scheme | MXFP8 |
| Quantized Size | 36740 MB |
| Task | Accuracy |
|---|---|
| gsm8k | 0.6088 |
| hellaswag | 0.6240 |
| mmlu | 0.8256 |
| mmlu_abstract_algebra | 0.6800 |
| mmlu_anatomy | 0.8444 |
| mmlu_astronomy | 0.9276 |
| mmlu_business_ethics | 0.8400 |
| mmlu_clinical_knowledge | 0.8943 |
| mmlu_college_biology | 0.9444 |
| mmlu_college_chemistry | 0.6200 |
| mmlu_college_computer_science | 0.7500 |
| mmlu_college_mathematics | 0.6800 |
| mmlu_college_medicine | 0.8671 |
| mmlu_college_physics | 0.7353 |
| mmlu_computer_security | 0.8300 |
| mmlu_conceptual_physics | 0.9489 |
| mmlu_econometrics | 0.7632 |
| mmlu_electrical_engineering | 0.8207 |
| mmlu_elementary_mathematics | 0.8016 |
| mmlu_formal_logic | 0.6746 |
| mmlu_global_facts | 0.5200 |
| mmlu_high_school_biology | 0.9548 |
| mmlu_high_school_chemistry | 0.8325 |
| mmlu_high_school_computer_science | 0.8900 |
| mmlu_high_school_european_history | 0.8000 |
| mmlu_high_school_geography | 0.9444 |
| mmlu_high_school_government_and_politics | 0.9793 |
| mmlu_high_school_macroeconomics | 0.9128 |
| mmlu_high_school_mathematics | 0.5852 |
| mmlu_high_school_microeconomics | 0.9664 |
| mmlu_high_school_physics | 0.8013 |
| mmlu_high_school_psychology | 0.9541 |
| mmlu_high_school_statistics | 0.8333 |
| mmlu_high_school_us_history | 0.9216 |
| mmlu_high_school_world_history | 0.8903 |
| mmlu_human_aging | 0.8206 |
| mmlu_human_sexuality | 0.9008 |
| mmlu_humanities | 0.7524 |
| mmlu_international_law | 0.9008 |
| mmlu_jurisprudence | 0.8981 |
| mmlu_logical_fallacies | 0.8896 |
| mmlu_machine_learning | 0.8036 |
| mmlu_management | 0.9223 |
| mmlu_marketing | 0.9359 |
| mmlu_medical_genetics | 0.9700 |
| mmlu_miscellaneous | 0.9413 |
| mmlu_moral_disputes | 0.8468 |
| mmlu_moral_scenarios | 0.6279 |
| mmlu_nutrition | 0.8889 |
| mmlu_other | 0.8651 |
| mmlu_philosophy | 0.8392 |
| mmlu_prehistory | 0.8981 |
| mmlu_professional_accounting | 0.7660 |
| mmlu_professional_law | 0.6610 |
| mmlu_professional_medicine | 0.9338 |
| mmlu_professional_psychology | 0.8922 |
| mmlu_public_relations | 0.7636 |
| mmlu_security_studies | 0.8041 |
| mmlu_social_sciences | 0.9084 |
| mmlu_sociology | 0.9303 |
| mmlu_stem | 0.8151 |
| mmlu_us_foreign_policy | 0.9400 |
| mmlu_virology | 0.5542 |
| mmlu_world_religions | 0.8889 |
| piqa | 0.8270 |
pip install auto-round1from transformers import AutoModelForCausalLM, AutoTokenizer
2
3model_name = "Qwen3.5-35B-A3B-AutoRound-MXFP8-ModelFree"
4
5# load the tokenizer and the model
6tokenizer = AutoTokenizer.from_pretrained(model_name)
7model = AutoModelForCausalLM.from_pretrained(model_name, torch_dtype="auto", device_map="auto")
8
9# prepare the model input
10prompt = "Write a quick sort algorithm."
11messages = [{"role": "user", "content": prompt}]
12text = tokenizer.apply_chat_template(
13 messages,
14 tokenize=False,
15 add_generation_prompt=True,
16)
17model_inputs = tokenizer([text], return_tensors="pt").to(model.device)
18
19# conduct text completion
20generated_ids = model.generate(**model_inputs, max_new_tokens=512)
21output_ids = generated_ids[0][len(model_inputs.input_ids[0]) :].tolist()
22
23content = tokenizer.decode(output_ids, skip_special_tokens=True)
24print("content:", content)1vllm serve Qwen3.5-35B-A3B-AutoRound-MXFP8-ModelFree \
2 --trust-remote-code \
3 --dtype bfloat16 \
4 --tensor_parallel_size 1@article{cheng2023optimize,
title={Optimize weight rounding via signed gradient descent for the quantization of llms},
author={Cheng, Wenhua and Zhang, Weiwei and Shen, Haihao and Cai, Yiyang and He, Xin and Lv, Kaokao and Liu, Yi},
journal={arXiv preprint arXiv:2309.05516},
year={2023}
}