Views
No views yet
lm-eval --model vllm --model_args pretrained=models/gptqmodel/Qwopus3.5-9B-v3.5-INT4-FOEM,tensor_parallel_size=1,gpu_memory_utilization=0.45 --tasks wikitext --batch_size 11@misc{jackrong_qwopus35_9b_v35,
2 title = {Qwopus3.5-9B-v3.5},
3 author = {Jackrong},
4 year = {2026},
5 publisher = {Hugging Face}
6}1@misc{qubitium2024gptqmodel,
2 author = {ModelCloud.ai and qubitium@modelcloud.ai},
3 title = {GPT-QModel},
4 publisher = {GitHub},
5 journal = {GitHub repository},
6 howpublished = {\url{https://github.com/modelcloud/gptqmodel}},
7 note = {Contact: qubitium@modelcloud.ai},
8 year = {2024},
9}1@inproceedings{zheng2026first,
2 title={First-order error matters: Accurate compensation for quantized large language models},
3 author={Zheng, Xingyu and Qin, Haotong and Li, Yuye and Chu, Haoran and Wang, Jiakai and Guo, Jinyang and Magno, Michele and Liu, Xianglong},
4 booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
5 volume={40},
6 number={34},
7 pages={28883--28891},
8 year={2026}
9}