Views
No views yet
8.0 or higher.nvcr.io/nvidia/pytorch:23.06-py3 image is runtime v12.1 but otherwise the same as the configuration above and has also been verified to work.1git clone https://github.com/mit-han-lab/llm-awq \
2&& cd llm-awq \
3&& git checkout f084f40bd996f3cf3a0633c1ad7d9d476c318aaa \
4&& pip install -e . \
5&& cd awq/kernels \
6&& python setup.py install1import time
2import torch
3from awq.quantize.quantizer import real_quantize_model_weight
4from transformers import AutoModelForCausalLM, AutoConfig, AutoTokenizer, TextStreamer
5from accelerate import init_empty_weights, load_checkpoint_and_dispatch
6from huggingface_hub import snapshot_download
7
8model_name = "abhinavkulkarni/VMware-open-llama-7b-open-instruct"
9
10# Config
11config = AutoConfig.from_pretrained(model_name, trust_remote_code=True)
12
13# Tokenizer
14try:
15 tokenizer = AutoTokenizer.from_pretrained(config.tokenizer_name, trust_remote_code=True)
16except:
17 tokenizer = AutoTokenizer.from_pretrained(model_name, use_fast=False, trust_remote_code=True)
18streamer = TextStreamer(tokenizer, skip_special_tokens=True)
19
20# Model
21w_bit = 4
22q_config = {
23 "zero_point": True,
24 "q_group_size": 128,
25}
26
27load_quant = snapshot_download(model_name)
28
29with init_empty_weights():
30 model = AutoModelForCausalLM.from_config(config=config,
31 torch_dtype=torch.float16, trust_remote_code=True)
32
33real_quantize_model_weight(model, w_bit=w_bit, q_config=q_config, init_only=True)
34model.tie_weights()
35
36model = load_checkpoint_and_dispatch(model, load_quant, device_map="balanced")
37
38# Inference
39prompt = f'''What is the difference between nuclear fusion and fission?
40###Response:'''
41
42input_ids = tokenizer(prompt, return_tensors='pt').input_ids.cuda()
43output = model.generate(
44 inputs=input_ids,
45 temperature=0.7,
46 max_new_tokens=512,
47 top_p=0.15,
48 top_k=0,
49 repetition_penalty=1.1,
50 eos_token_id=tokenizer.eos_token_id,
51 streamer=streamer)| Task | Version | Metric | Value | Stderr | |
|---|---|---|---|---|---|
| wikitext | 1 | word_perplexity | 11.7531 | ||
| byte_perplexity | 1.5853 | ||||
| bits_per_byte | 0.6648 |
| Task | Version | Metric | Value | Stderr | |
|---|---|---|---|---|---|
| wikitext | 1 | word_perplexity | 12.1840 | ||
| byte_perplexity | 1.5961 | ||||
| bits_per_byte | 0.6745 |
@software{openlm2023openllama,
author = {Geng, Xinyang and Liu, Hao},
title = {OpenLLaMA: An Open Reproduction of LLaMA},
month = May,
year = 2023,
url = {https://github.com/openlm-research/open_llama}
}@software{together2023redpajama,
author = {Together Computer},
title = {RedPajama-Data: An Open Source Recipe to Reproduce LLaMA training dataset},
month = April,
year = 2023,
url = {https://github.com/togethercomputer/RedPajama-Data}
}@article{touvron2023llama,
title={Llama: Open and efficient foundation language models},
author={Touvron, Hugo and Lavril, Thibaut and Izacard, Gautier and Martinet, Xavier and Lachaux, Marie-Anne and Lacroix, Timoth{\'e}e and Rozi{\`e}re, Baptiste and Goyal, Naman and Hambro, Eric and Azhar, Faisal and others},
journal={arXiv preprint arXiv:2302.13971},
year={2023}
}@article{lin2023awq,
title={AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration},
author={Lin, Ji and Tang, Jiaming and Tang, Haotian and Yang, Shang and Dang, Xingyu and Han, Song},
journal={arXiv},
year={2023}
}