Views
No views yet
robinshao/safe_huihui_qwen3_5_4bsafe_huihui_qwen3_5_4b_mxfp4_gguf.gguf: GGUF produced with the local patched llama.cpp toolchainllama.cpp.src-patched.zip: current uncompiled patched source treellama.cpp.build-local-bin.zip: compiled binaries matching the source treellama.cpp.build-windows-cuda-local-bin.zip: additional compiled patched llama.cpp binariesbuild_result.json: build, quantization, and smoke-test summary1{
2 "arch": "qwen35",
3 "tensor_count": 426,
4 "type_counts": {
5 "Q8_0": 1,
6 "F32": 177,
7 "MXFP4": 248
8 },
9 "converted_preview": [
10 {
11 "index": 1,
12 "name": "token_embd.weight",
13 "from": "BF16",
14 "to": "Q8_0",
15 "bytes": 675430400
16 },
17 {
18 "index": 2,
19 "name": "blk.0.attn_norm.weight",
20 "from": "F32",
21 "to": "F32",
22 "bytes": 10240
23 },
24 {
25 "index": 3,
26 "name": "blk.0.ssm_a",
27 "from": "F32",
28 "to": "F32",
29 "bytes": 128
30 },
31 {
32 "index": 4,
33 "name": "blk.0.ssm_conv1d.weight",
34 "from": "F32",
35 "to": "F32",
36 "bytes": 131072
37 },
38 {
39 "index": 5,
40 "name": "blk.0.ssm_dt.bias",
41 "from": "F32",
42 "to": "F32",
43 "bytes": 128
44 },
45 {
46 "index": 6,
47 "name": "blk.0.ssm_alpha.weight",
48 "from": "BF16",
49 "to": "MXFP4",
50 "bytes": 43520
51 },
52 {
53 "index": 7,
54 "name": "blk.0.ssm_beta.weight",
55 "from": "BF16",
56 "to": "MXFP4",
57 "bytes": 43520
58 },
59 {
60 "index": 8,
61 "name": "blk.0.attn_qkv.weight",
62 "from": "BF16",
63 "to": "MXFP4",
64 "bytes": 11141120
65 },
66 {
67 "index": 9,
68 "name": "blk.0.attn_gate.weight",
69 "from": "BF16",
70 "to": "MXFP4",
71 "bytes": 5570560
72 },
73 {
74 "index": 10,
75 "name": "blk.0.ssm_norm.weight",
76 "from": "F32",
77 "to": "F32",
78 "bytes": 512
79 },
80 {
81 "index": 11,
82 "name": "blk.0.ssm_out.weight",
83 "from": "BF16",
84 "to": "MXFP4",
85 "bytes": 5570560
86 },
87 {
88 "index": 12,
89 "name": "blk.0.ffn_down.weight",
90 "from": "BF16",
91 "to": "MXFP4",
92 "bytes": 12533760
93 },
94 {
95 "index": 13,
96 "name": "blk.0.ffn_gate.weight",
97 "from": "BF16",
98 "to": "MXFP4",
99 "bytes": 12533760
100 },
101 {
102 "index": 14,
103 "name": "blk.0.ffn_up.weight",
104 "from": "BF16",
105 "to": "MXFP4",
106 "bytes": 12533760
107 },
108 {
109 "index": 15,
110 "name": "blk.0.post_attention_norm.weight",
111 "from": "F32",
112 "to": "F32",
113 "bytes": 10240
114 },
115 {
116 "index": 16,
117 "name": "blk.1.attn_norm.weight",
118 "from": "F32",
119 "to": "F32",
120 "bytes": 10240
121 },
122 {
123 "index": 17,
124 "name": "blk.1.ssm_a",
125 "from": "F32",
126 "to": "F32",
127 "bytes": 128
128 },
129 {
130 "index": 18,
131 "name": "blk.1.ssm_conv1d.weight",
132 "from": "F32",
133 "to": "F32",
134 "bytes": 131072
135 },
136 {
137 "index": 19,
138 "name": "blk.1.ssm_dt.bias",
139 "from": "F32",
140 "to": "F32",
141 "bytes": 128
142 },
143 {
144 "index": 20,
145 "name": "blk.1.ssm_alpha.weight",
146 "from": "BF16",
147 "to": "MXFP4",
148 "bytes": 43520
149 },
150 {
151 "index": 21,
152 "name": "blk.1.ssm_beta.weight",
153 "from": "BF16",
154 "to": "MXFP4",
155 "bytes": 43520
156 },
157 {
158 "index": 22,
159 "name": "blk.1.attn_qkv.weight",
160 "from": "BF16",
161 "to": "MXFP4",
162 "bytes": 11141120
163 },
164 {
165 "index": 23,
166 "name": "blk.1.attn_gate.weight",
167 "from": "BF16",
168 "to": "MXFP4",
169 "bytes": 5570560
170 },
171 {
172 "index": 24,
173 "name": "blk.1.ssm_norm.weight",
174 "from": "F32",
175 "to": "F32",
176 "bytes": 512
177 }
178 ],
179 "output_bytes": 2586323264
180}1{
2 "ok": true,
3 "returncode": 0,
4 "timed_out": false,
5 "load_seen": true,
6 "elapsed_sec": 3.02,
7 "stdout_tail": "\nLoading model... \n\n\n▄▄ ▄▄\n██ ██\n██ ██ ▀▀█▄ ███▄███▄ ▀▀█▄ ▄████ ████▄ ████▄\n██ ██ ▄█▀██ ██ ██ ██ ▄█▀██ ██ ██ ██ ██ ██\n██ ██ ▀█▄██ ██ ██ ██ ▀█▄██ ██ ▀████ ████▀ ████▀\n ██ ██\n ▀▀ ▀▀\n\nbuild : b0-unknown\nmodel : robinshao__safe_huihui_qwen3_5_4b-mxfp4.gguf\nmodalities : text\n\navailable commands:\n /exit or Ctrl+C stop or exit\n /regen regenerate the last response\n /clear clear the chat history\n /read <file> add a text file\n /glob <pattern> add text files using globbing pattern\n\n\n> OK\n\n[Start thinking]\nOkay\n\n[ Prompt: 65.0 t/s | Generation: 1000000.0 t/s ]\n\nExiting...\n",
8 "stderr_tail": "warning: no usable GPU found, --gpu-layers option will be ignored\nwarning: one possible reason is that llama.cpp was compiled without GPU support\nwarning: consult docs/build.md for compilation instructions\n",
9 "ctx_size": 256,
10 "ngl": 0,
11 "predict_tokens": 1,
12 "timeout_sec": 120,
13 "gpu_layers_forced": true,
14 "stdout_log": "/data/cz_qwen_mxfp4_work/runs/robinshao__safe_huihui_qwen3_5_4b/smoke_logs/robinshao__safe_huihui_qwen3_5_4b-mxfp4.out.log",
15 "stderr_log": "/data/cz_qwen_mxfp4_work/runs/robinshao__safe_huihui_qwen3_5_4b/smoke_logs/robinshao__safe_huihui_qwen3_5_4b-mxfp4.err.log"
16}