pip install -U -q keras-hub
pip install -U -q keras| Preset name | Parameters | Description |
|---|---|---|
| mixtral_8_7b_en | 7B | 32-layer Mixtral MoE model with 7 billion active parameters and 8 experts per MoE layer. |
| mixtral_8_instruct_7b_en | 7B | Instruction fine-tuned 32-layer Mixtral MoE model with 7 billion active parameters and 8 experts per MoE layer. |
1
2import keras
3import keras_hub
4import numpy as np
5
6# Basic text generation
7mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset("mixtral_8_instruct_7b_en")
8mixtral_lm.generate("[INST] What is Keras? [/INST]", max_length=500)
9
10# Generate with batched prompts
11mixtral_lm.generate([
12 "[INST] What is Keras? [/INST]",
13 "[INST] Give me your best brownie recipe. [/INST]"
14], max_length=500)
15
16# Using different sampling strategies
17mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset("mixtral_8_instruct_7b_en")
18# Greedy sampling
19mixtral_lm.compile(sampler="greedy")
20mixtral_lm.generate("I want to say", max_length=30)
21
22# Beam search
23mixtral_lm.compile(
24 sampler=keras_hub.samplers.BeamSampler(
25 num_beams=2,
26 top_k_experts=2, # MoE-specific: number of experts to use per token
27 )
28)
29mixtral_lm.generate("I want to say", max_length=30)
30
31# Generate without preprocessing
32prompt = {
33 "token_ids": np.array([[1, 315, 947, 298, 1315, 0, 0, 0, 0, 0]] * 2),
34 "padding_mask": np.array([[1, 1, 1, 1, 1, 0, 0, 0, 0, 0]] * 2),
35}
36
37mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset(
38 "mixtral_8_instruct_7b_en",
39 preprocessor=None,
40 dtype="bfloat16"
41)
42mixtral_lm.generate(
43 prompt,
44 num_experts=8, # Total number of experts per layer
45 top_k_experts=2, # Number of experts to use per token
46 router_aux_loss_coef=0.02 # Router auxiliary loss coefficient
47)
48
49# Training on a single batch
50features = ["The quick brown fox jumped.", "I forgot my homework."]
51mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset(
52 "mixtral_8_instruct_7b_en",
53 dtype="bfloat16"
54)
55mixtral_lm.fit(
56 x=features,
57 batch_size=2,
58 router_aux_loss_coef=0.02 # MoE-specific: router training loss
59)
60
61# Training without preprocessing
62x = {
63 "token_ids": np.array([[1, 315, 947, 298, 1315, 369, 315, 837, 0, 0]] * 2),
64 "padding_mask": np.array([[1, 1, 1, 1, 1, 1, 1, 1, 0, 0]] * 2),
65}
66y = np.array([[315, 947, 298, 1315, 369, 315, 837, 0, 0, 0]] * 2)
67sw = np.array([[1, 1, 1, 1, 1, 1, 1, 0, 0, 0]] * 2)
68
69mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset(
70 "mixtral_8_instruct_7b_en",
71 preprocessor=None,
72 dtype="bfloat16"
73)
74mixtral_lm.fit(
75 x=x,
76 y=y,
77 sample_weight=sw,
78 batch_size=2,
79 router_aux_loss_coef=0.02
80)1
2import keras
3import keras_hub
4import numpy as np
5
6# Basic text generation
7mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset("hf://keras/mixtral_8_instruct_7b_en")
8mixtral_lm.generate("[INST] What is Keras? [/INST]", max_length=500)
9
10# Generate with batched prompts
11mixtral_lm.generate([
12 "[INST] What is Keras? [/INST]",
13 "[INST] Give me your best brownie recipe. [/INST]"
14], max_length=500)
15
16# Using different sampling strategies
17mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset("hf://keras/mixtral_8_instruct_7b_en")
18# Greedy sampling
19mixtral_lm.compile(sampler="greedy")
20mixtral_lm.generate("I want to say", max_length=30)
21
22# Beam search
23mixtral_lm.compile(
24 sampler=keras_hub.samplers.BeamSampler(
25 num_beams=2,
26 top_k_experts=2, # MoE-specific: number of experts to use per token
27 )
28)
29mixtral_lm.generate("I want to say", max_length=30)
30
31# Generate without preprocessing
32prompt = {
33 "token_ids": np.array([[1, 315, 947, 298, 1315, 0, 0, 0, 0, 0]] * 2),
34 "padding_mask": np.array([[1, 1, 1, 1, 1, 0, 0, 0, 0, 0]] * 2),
35}
36
37mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset(
38 "hf://keras/mixtral_8_instruct_7b_en",
39 preprocessor=None,
40 dtype="bfloat16"
41)
42mixtral_lm.generate(
43 prompt,
44 num_experts=8, # Total number of experts per layer
45 top_k_experts=2, # Number of experts to use per token
46 router_aux_loss_coef=0.02 # Router auxiliary loss coefficient
47)
48
49# Training on a single batch
50features = ["The quick brown fox jumped.", "I forgot my homework."]
51mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset(
52 "hf://keras/mixtral_8_instruct_7b_en",
53 dtype="bfloat16"
54)
55mixtral_lm.fit(
56 x=features,
57 batch_size=2,
58 router_aux_loss_coef=0.02 # MoE-specific: router training loss
59)
60
61# Training without preprocessing
62x = {
63 "token_ids": np.array([[1, 315, 947, 298, 1315, 369, 315, 837, 0, 0]] * 2),
64 "padding_mask": np.array([[1, 1, 1, 1, 1, 1, 1, 1, 0, 0]] * 2),
65}
66y = np.array([[315, 947, 298, 1315, 369, 315, 837, 0, 0, 0]] * 2)
67sw = np.array([[1, 1, 1, 1, 1, 1, 1, 0, 0, 0]] * 2)
68
69mixtral_lm = keras_hub.models.MixtralCausalLM.from_preset(
70 "hf://keras/mixtral_8_instruct_7b_en",
71 preprocessor=None,
72 dtype="bfloat16"
73)
74mixtral_lm.fit(
75 x=x,
76 y=y,
77 sample_weight=sw,
78 batch_size=2,
79 router_aux_loss_coef=0.02
80)