Views
No views yet

"Base Model": "meta-llama/Meta-Llama-3-8B-Instruct",
"Reasoning LoRA-Expert": "abacusai/Llama-3-Smaug-8B,
"Function Calling LoRA-Expert": "hiieu/Meta-Llama-3-8B-Instruct-function-calling-json-mode",
"Python LoRA-Expert": "rombodawg/Llama-3-8B-Instruct-Coder",
"SQL LoRA-Expert": "defog/llama-3-sqlcoder-8b",
"German LoRA-Expert": "VAGOsolutions/Llama-3-SauerkrautLM-8b-Instruct"from transformers import AutoModelForCausalLM
device = "cuda:0" ## Setup "cuda:0" if NVIDIA, "mps" if on Mac
# Load the model and config:
model = AutoModelForCausalLM.from_pretrained("./kraken_model", trust_remote_code=True)messages = [
{'role': 'system', 'content': '"You are a helpful AI Assistant'},
{'role': 'user', 'content': "Find the mass percentage of Ba in BaO"}
]
tokenizer = model.tokenizer
input_text = tokenizer.apply_chat_template(messages, tokenize=False)
input_ids = tokenizer(input_text, return_tensors="pt").input_ids.to(device)
output_ids = model.generate(input_ids, max_length=250)
print(tokenizer.decode(output_ids[0], skip_special_tokens=True))functions_metadata = [
{
"type": "function",
"function": {
"name": "get_temperature",
"description": "get temperature of a city",
"parameters": {
"type": "object",
"properties": {
"city": {
"type": "string",
"description": "name"
}
},
"required": [
"city"
]
}
}
}
]
messages = [
{ "role": "system", "content": f"""You are a helpful assistant with access to the following functions: \n {str(functions_metadata)}\n\nTo use these functions respond with:\n<functioncall> {{ "name": "function_name", "arguments": {{ "arg_1": "value_1", "arg_1": "value_1", ... }} }} </functioncall>\n\nEdge cases you must handle:\n - If there are no functions that match the user request, you will respond politely that you cannot help."""},
{ "role": "user", "content": """<function_response> {"temperature": 12} </function_response>"""}
]
tokenizer = model.tokenizer
input_text = tokenizer.apply_chat_template(messages, tokenize=False)
input_ids = tokenizer(input_text, return_tensors="pt").input_ids.to("cuda:0")
output_ids = model.generate(input_ids ,temperature=0.1, do_sample=True, top_p=0.9,top_k=20, max_length=500)
print(tokenizer.decode(output_ids[0], skip_special_tokens=True))messages = [
{'role': 'system', 'content': ''},
{'role': 'user', 'content': """Create a python function to calculate the sum of a sequence of integers.
[1, 2, 3, 4, 5]"""}
]
tokenizer = model.tokenizer
input_text = tokenizer.apply_chat_template(messages, tokenize=False)
print(input_text)
input_ids = tokenizer(input_text, return_tensors="pt").input_ids.to("cuda:0")
output_ids = model.generate(input_ids ,temperature=0.6, do_sample=True, top_p=0.9,top_k=20, max_length=400)
print(tokenizer.decode(output_ids[0], skip_special_tokens=True))messages = [
{'role': 'system', 'content': 'You are a helpul AI assistant.'},
{'role': 'user', 'content': """Generate a SQL query to answer this question: What is the total volume of timber sold by each salesperson, sorted by salesperson?
DDL statements:
CREATE TABLE salesperson (salesperson_id INT, name TEXT, region TEXT); INSERT INTO salesperson (salesperson_id, name, region) VALUES (1, 'John Doe', 'North'), (2, 'Jane Smith', 'South'); CREATE TABLE timber_sales (sales_id INT, salesperson_id INT, volume REAL, sale_date DATE); INSERT INTO timber_sales (sales_id, salesperson_id, volume, sale_date) VALUES (1, 1, 120, '2021-01-01'), (2, 1, 150, '2021-02-01'), (3, 2, 180, '2021-01-01');"""}
]
tokenizer = model.tokenizer
input_text = tokenizer.apply_chat_template(messages, tokenize=False)
input_ids = tokenizer(input_text, return_tensors="pt").input_ids.to(device)
output_ids = model.generate(input_ids ,temperature=0.6, do_sample=True, top_p=0.9,top_k=20, max_length=500)
print(tokenizer.decode(output_ids[0], skip_special_tokens=True))messages = [
{'role': 'system', 'content': 'Du bist ein freundlicher und hilfreicher deutscher KI-Assistent'},
{'role': 'user', 'content': "Ich hoffe es geht dir gut?"}
]
tokenizer = model.tokenizer
input_text = tokenizer.apply_chat_template(messages, tokenize=False)
input_ids = tokenizer(input_text, return_tensors="pt").input_ids.to("cuda:0")
output_ids = model.generate(input_ids, max_length=150)
print(tokenizer.decode(output_ids[0], skip_special_tokens=True)) # Switch to a LoRA-Adapter which fits to your Base Model
"lora_adapters": {
"lora_expert1": "Llama-3-Smaug-8B-adapter",
"lora_expert2": "Meta-Llama-3-8B-Instruct-function-calling-json-mode-adapter",
"lora_expert3": "Llama-3-8B-Instruct-Coder-adapter",
"lora_expert4": "llama-3-sqlcoder-8b-adapter",
"lora_expert5": "Llama-3-SauerkrautLM-8b-Instruct-adapter"
},
"model_type": "kraken",
"models": {
"base": "meta-llama/Meta-Llama-3-8B-Instruct"
},
# Currently supported: "4bit" and "8bit"
"quantization": {
"base": null
},
"router": "../kraken/kraken_router",
"tokenizers": {
"lora_expert1": "Llama-3-Smaug-8B-adapter",
"lora_expert2": "Meta-Llama-3-8B-Instruct-function-calling-json-mode-adapter",
"lora_expert3": "Llama-3-8B-Instruct-Coder-adapter",
"lora_expert4": "llama-3-sqlcoder-8b-adapter",
"lora_expert5": "Llama-3-SauerkrautLM-8b-Instruct-adapter"
}
},
"model_type": "kraken",
"torch_dtype": "bfloat16",
"transformers_version": "4.41.1"
}