Views
No views yet
lm-eval --model vllm --model_args pretrained=models/gptqmodel/Qwen3.5-9B-GLM5.1-Distill-v1-INT8-FOEM,tensor_parallel_size=1,gpu_memory_utilization=0.45 --tasks wikitext --batch_size 11@misc{jackrong_qwen35_27b_v3
2 title = {Jackrong/Qwen3.5-9B-GLM5.1-Distill-v1},
3 author = {Jackrong},
4 year = {2026},
5 publisher = {Hugging Face},
6 howpublished = {\url{https://huggingface.co/Jackrong/Qwen3.5-9B-GLM5.1-Distill-v1}}
7}1@misc{qubitium2024gptqmodel,
2 author = {ModelCloud.ai and qubitium@modelcloud.ai},
3 title = {GPT-QModel},
4 publisher = {GitHub},
5 journal = {GitHub repository},
6 howpublished = {\url{https://github.com/modelcloud/gptqmodel}},
7 note = {Contact: qubitium@modelcloud.ai},
8 year = {2024},
9}1@inproceedings{zheng2026first,
2 title={First-order error matters: Accurate compensation for quantized large language models},
3 author={Zheng, Xingyu and Qin, Haotong and Li, Yuye and Chu, Haoran and Wang, Jiakai and Guo, Jinyang and Magno, Michele and Liu, Xianglong},
4 booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
5 volume={40},
6 number={34},
7 pages={28883--28891},
8 year={2026}
9}