Views
No views yet
robinshao/sese_qwen36_35b_a3b_hereticsese_qwen36_35b_a3b_heretic_mxfp4_gguf.gguf: GGUF produced with the local patched llama.cpp toolchainllama.cpp.src-patched.zip: current uncompiled patched source treellama.cpp.build-local-bin.zip: compiled binaries matching the source treellama.cpp.build-windows-cuda-local-bin.zip: additional compiled patched llama.cpp binariesbuild_result.json: build, quantization, and smoke-test summary1{
2 "arch": "qwen35moe",
3 "tensor_count": 733,
4 "type_counts": {
5 "Q8_0": 2,
6 "F32": 301,
7 "MXFP4": 430
8 },
9 "converted_preview": [
10 {
11 "index": 1,
12 "name": "output.weight",
13 "from": "BF16",
14 "to": "Q8_0",
15 "bytes": 540344320
16 },
17 {
18 "index": 2,
19 "name": "token_embd.weight",
20 "from": "BF16",
21 "to": "Q8_0",
22 "bytes": 540344320
23 },
24 {
25 "index": 3,
26 "name": "blk.0.attn_norm.weight",
27 "from": "F32",
28 "to": "F32",
29 "bytes": 8192
30 },
31 {
32 "index": 4,
33 "name": "blk.0.ssm_a",
34 "from": "F32",
35 "to": "F32",
36 "bytes": 128
37 },
38 {
39 "index": 5,
40 "name": "blk.0.ssm_conv1d.weight",
41 "from": "F32",
42 "to": "F32",
43 "bytes": 131072
44 },
45 {
46 "index": 6,
47 "name": "blk.0.ssm_dt.bias",
48 "from": "F32",
49 "to": "F32",
50 "bytes": 128
51 },
52 {
53 "index": 7,
54 "name": "blk.0.ssm_alpha.weight",
55 "from": "BF16",
56 "to": "MXFP4",
57 "bytes": 34816
58 },
59 {
60 "index": 8,
61 "name": "blk.0.ssm_beta.weight",
62 "from": "BF16",
63 "to": "MXFP4",
64 "bytes": 34816
65 },
66 {
67 "index": 9,
68 "name": "blk.0.attn_qkv.weight",
69 "from": "BF16",
70 "to": "MXFP4",
71 "bytes": 8912896
72 },
73 {
74 "index": 10,
75 "name": "blk.0.attn_gate.weight",
76 "from": "BF16",
77 "to": "MXFP4",
78 "bytes": 4456448
79 },
80 {
81 "index": 11,
82 "name": "blk.0.ssm_norm.weight",
83 "from": "F32",
84 "to": "F32",
85 "bytes": 512
86 },
87 {
88 "index": 12,
89 "name": "blk.0.ssm_out.weight",
90 "from": "BF16",
91 "to": "MXFP4",
92 "bytes": 4456448
93 },
94 {
95 "index": 13,
96 "name": "blk.0.ffn_down_exps.weight",
97 "from": "BF16",
98 "to": "MXFP4",
99 "bytes": 142606336
100 },
101 {
102 "index": 14,
103 "name": "blk.0.ffn_gate_exps.weight",
104 "from": "BF16",
105 "to": "MXFP4",
106 "bytes": 142606336
107 },
108 {
109 "index": 15,
110 "name": "blk.0.ffn_up_exps.weight",
111 "from": "BF16",
112 "to": "MXFP4",
113 "bytes": 142606336
114 },
115 {
116 "index": 16,
117 "name": "blk.0.ffn_gate_inp.weight",
118 "from": "F32",
119 "to": "F32",
120 "bytes": 2097152
121 },
122 {
123 "index": 17,
124 "name": "blk.0.ffn_down_shexp.weight",
125 "from": "BF16",
126 "to": "MXFP4",
127 "bytes": 557056
128 },
129 {
130 "index": 18,
131 "name": "blk.0.ffn_gate_shexp.weight",
132 "from": "BF16",
133 "to": "MXFP4",
134 "bytes": 557056
135 },
136 {
137 "index": 19,
138 "name": "blk.0.ffn_up_shexp.weight",
139 "from": "BF16",
140 "to": "MXFP4",
141 "bytes": 557056
142 },
143 {
144 "index": 20,
145 "name": "blk.0.ffn_gate_inp_shexp.weight",
146 "from": "F32",
147 "to": "F32",
148 "bytes": 8192
149 },
150 {
151 "index": 21,
152 "name": "blk.0.post_attention_norm.weight",
153 "from": "F32",
154 "to": "F32",
155 "bytes": 8192
156 },
157 {
158 "index": 22,
159 "name": "blk.1.attn_norm.weight",
160 "from": "F32",
161 "to": "F32",
162 "bytes": 8192
163 },
164 {
165 "index": 23,
166 "name": "blk.1.ssm_a",
167 "from": "F32",
168 "to": "F32",
169 "bytes": 128
170 },
171 {
172 "index": 24,
173 "name": "blk.1.ssm_conv1d.weight",
174 "from": "F32",
175 "to": "F32",
176 "bytes": 131072
177 }
178 ],
179 "output_bytes": 19041835072
180}1{
2 "ok": true,
3 "returncode": 0,
4 "timed_out": false,
5 "load_seen": false,
6 "elapsed_sec": 4.98,
7 "stdout_tail": ",\n\n",
8 "stderr_tail": "m_params: warming up the model with an empty run - please wait ... (--no-warmup to disable)\n0.04.660.350 I llama_completion: llama threadpool init, n_threads = 32\n0.04.663.327 I \n0.04.663.408 I system_info: n_threads = 32 (n_threads_batch = 32) / 64 | CUDA : ARCHS = 1200 | USE_GRAPHS = 1 | PEER_MAX_BATCH_SIZE = 128 | BLACKWELL_NATIVE_FP4 = 1 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | LLAMAFILE = 1 | OPENMP = 1 | REPACK = 1 | \n0.04.663.410 I \n0.04.663.522 I sampler seed: 1307290118\n0.04.663.533 I sampler params: \n\trepeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000\n\tdry_multiplier = 0.000, dry_base = 1.750, dry_allowed_length = 2, dry_penalty_last_n = -1\n\ttop_k = 20, top_p = 0.950, min_p = 0.050, xtc_probability = 0.000, xtc_threshold = 0.100, typical_p = 1.000, top_n_sigma = -1.000, temp = 0.000\n\tmirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000, adaptive_target = -1.000, adaptive_decay = 0.900\n0.04.663.539 I sampler chain: logits -> ?penalties -> ?dry -> ?top-n-sigma -> top-k -> ?typical -> top-p -> min-p -> ?xtc -> temp-ext -> dist \n0.04.663.540 I generate: n_ctx = 256, n_batch = 2048, n_predict = 1, n_keep = 0\n0.04.663.540 I \n0.04.675.617 I common_perf_print: sampling time = 1.72 ms\n0.04.675.619 I common_perf_print: samplers time = 0.08 ms / 2 tokens\n0.04.675.622 I common_perf_print: load time = 3409.80 ms\n0.04.675.623 I common_perf_print: prompt eval time = 0.00 ms / 1 tokens ( 0.00 ms per token, inf tokens per second)\n0.04.675.624 I common_perf_print: eval time = 10.33 ms / 1 runs ( 10.33 ms per token, 96.81 tokens per second)\n0.04.675.624 I common_perf_print: total time = 15.27 ms / 2 tokens\n0.04.675.625 I common_perf_print: unaccounted time = 3.23 ms / 21.1 % (total - sampling - prompt eval - eval) / (total)\n0.04.675.625 I common_perf_print: graphs reused = 0\n",
9 "ctx_size": 256,
10 "ngl": 999,
11 "predict_tokens": 1,
12 "timeout_sec": 300,
13 "gpu_layers_forced": true,
14 "stdout_log": "/data/cz_qwen_mxfp4_work/runs/robinshao__sese_qwen36_35b_a3b_heretic/smoke_logs/robinshao__sese_qwen36_35b_a3b_heretic-mxfp4.out.log",
15 "stderr_log": "/data/cz_qwen_mxfp4_work/runs/robinshao__sese_qwen36_35b_a3b_heretic/smoke_logs/robinshao__sese_qwen36_35b_a3b_heretic-mxfp4.err.log"
16}