Views
No views yet

<|prompt|>{prompt}<|endoftext|><|answer|>git clone https://github.com/cmp-nct/ggllm.cpp
cd ggllm.cpp
rm -rf build && mkdir build && cd build && cmake -DGGML_CUBLAS=1 .. && cmake --build . --config Releasebin/falcon_main just like you would use llama.cpp. For example:bin/falcon_main -t 8 -ngl 100 -b 1 -m h2ogpt-gm-oasst1-en-2048-falcon-7b-v3.ggccv1.q4_0.bin -enc -p "write a story about llamas"-enc should automatically use the right prompt template for the model, so you can just enter your desired prompt.-ngl 100 regardles of your VRAM, as it will automatically detect how much VRAM is available to be used.-t 8 (the number of CPU cores to use) according to what performs best on your system. Do not exceed the number of physical CPU cores you have.-b 1 reduces batch size to 1. This slightly lowers prompt evaluation time, but frees up VRAM to load more of the model on to your GPU. If you find prompt evaluation too slow and have enough spare VRAM, you can remove this parameter.| Name | Quant method | Bits | Size | Max RAM required | Use case |
|---|---|---|---|---|---|
| h2ogpt-gm-oasst1-en-2048-falcon-7b-v3.ggccv1.q4_0.bin | q4_0 | 4 | 4.06 GB | 6.56 GB | Original quant method, 4-bit. |
| h2ogpt-gm-oasst1-en-2048-falcon-7b-v3.ggccv1.q4_1.bin | q4_1 | 4 | 4.51 GB | 7.01 GB | Original quant method, 4-bit. Higher accuracy than q4_0 but not as high as q5_0. However has quicker inference than q5 models. |
| h2ogpt-gm-oasst1-en-2048-falcon-7b-v3.ggccv1.q5_0.bin | q5_0 | 5 | 4.96 GB | 7.46 GB | Original quant method, 5-bit. Higher accuracy, higher resource usage and slower inference. |
| h2ogpt-gm-oasst1-en-2048-falcon-7b-v3.ggccv1.q5_1.bin | q5_1 | 5 | 5.41 GB | 7.91 GB | Original quant method, 5-bit. Even higher accuracy, resource usage and slower inference. |
| h2ogpt-gm-oasst1-en-2048-falcon-7b-v3.ggccv1.q8_0.bin | q8_0 | 8 | 7.67 GB | 10.17 GB | Original quant method, 8-bit. Almost indistinguishable from float16. High resource use and slow. Not recommended for most users. |
| Note: the above RAM figures assume no GPU offloading. If layers are offloaded to the GPU, this will reduce RAM usage and use VRAM instead. |
transformers library on a machine with GPUs, first make sure you have the transformers, accelerate, torch and einops libraries installed.1pip install transformers==4.29.2
2pip install accelerate==0.19.0
3pip install torch==2.0.0
4pip install einops==0.6.11import torch
2from transformers import AutoTokenizer, pipeline
3
4
5tokenizer = AutoTokenizer.from_pretrained(
6 "h2oai/h2ogpt-gm-oasst1-en-2048-falcon-7b-v3",
7 use_fast=False,
8 padding_side="left",
9 trust_remote_code=True,
10)
11
12generate_text = pipeline(
13 model="h2oai/h2ogpt-gm-oasst1-en-2048-falcon-7b-v3",
14 tokenizer=tokenizer,
15 torch_dtype=torch.float16,
16 trust_remote_code=True,
17 use_fast=False,
18 device_map={"": "cuda:0"},
19)
20
21res = generate_text(
22 "Why is drinking water so healthy?",
23 min_new_tokens=2,
24 max_new_tokens=1024,
25 do_sample=False,
26 num_beams=1,
27 temperature=float(0.3),
28 repetition_penalty=float(1.2),
29 renormalize_logits=True
30)
31print(res[0]["generated_text"])print(generate_text.preprocess("Why is drinking water so healthy?")["prompt_text"])<|prompt|>Why is drinking water so healthy?<|endoftext|><|answer|>1import torch
2from h2oai_pipeline import H2OTextGenerationPipeline
3from transformers import AutoModelForCausalLM, AutoTokenizer
4
5tokenizer = AutoTokenizer.from_pretrained(
6 "h2oai/h2ogpt-gm-oasst1-en-2048-falcon-7b-v3",
7 use_fast=False,
8 padding_side="left",
9 trust_remote_code=True,
10)
11model = AutoModelForCausalLM.from_pretrained(
12 "h2oai/h2ogpt-gm-oasst1-en-2048-falcon-7b-v3",
13 torch_dtype=torch.float16,
14 device_map={"": "cuda:0"},
15 trust_remote_code=True,
16)
17generate_text = H2OTextGenerationPipeline(model=model, tokenizer=tokenizer)
18
19res = generate_text(
20 "Why is drinking water so healthy?",
21 min_new_tokens=2,
22 max_new_tokens=1024,
23 do_sample=False,
24 num_beams=1,
25 temperature=float(0.3),
26 repetition_penalty=float(1.2),
27 renormalize_logits=True
28)
29print(res[0]["generated_text"])1from transformers import AutoModelForCausalLM, AutoTokenizer
2
3model_name = "h2oai/h2ogpt-gm-oasst1-en-2048-falcon-7b-v3" # either local folder or huggingface model name
4# Important: The prompt needs to be in the same format the model was trained with.
5# You can find an example prompt in the experiment logs.
6prompt = "<|prompt|>How are you?<|endoftext|><|answer|>"
7
8tokenizer = AutoTokenizer.from_pretrained(
9 model_name,
10 use_fast=False,
11 trust_remote_code=True,
12)
13model = AutoModelForCausalLM.from_pretrained(
14 model_name,
15 torch_dtype=torch.float16,
16 device_map={"": "cuda:0"},
17 trust_remote_code=True,
18)
19model.cuda().eval()
20inputs = tokenizer(prompt, return_tensors="pt", add_special_tokens=False).to("cuda")
21
22# generate configuration can be modified to your needs
23tokens = model.generate(
24 **inputs,
25 min_new_tokens=2,
26 max_new_tokens=1024,
27 do_sample=False,
28 num_beams=1,
29 temperature=float(0.3),
30 repetition_penalty=float(1.2),
31 renormalize_logits=True
32)[0]
33
34tokens = tokens[inputs["input_ids"].shape[1]:]
35answer = tokenizer.decode(tokens, skip_special_tokens=True)
36print(answer)RWForCausalLM(
(transformer): RWModel(
(word_embeddings): Embedding(65024, 4544)
(h): ModuleList(
(0-31): 32 x DecoderLayer(
(input_layernorm): LayerNorm((4544,), eps=1e-05, elementwise_affine=True)
(self_attention): Attention(
(maybe_rotary): RotaryEmbedding()
(query_key_value): Linear(in_features=4544, out_features=4672, bias=False)
(dense): Linear(in_features=4544, out_features=4544, bias=False)
(attention_dropout): Dropout(p=0.0, inplace=False)
)
(mlp): MLP(
(dense_h_to_4h): Linear(in_features=4544, out_features=18176, bias=False)
(act): GELU(approximate='none')
(dense_4h_to_h): Linear(in_features=18176, out_features=4544, bias=False)
)
)
)
(ln_f): LayerNorm((4544,), eps=1e-05, elementwise_affine=True)
)
(lm_head): Linear(in_features=4544, out_features=65024, bias=False)
)