Views
No views yet
Qwen/Qwen3.8-27B. It is not a fine-tune,
merge, ablation, alignment change, or chat-template modification. The source
weights are pinned to commit 1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0.Qwen3_5ForConditionalGeneration / qwen3_5 as
its internal architecture identifier. That string does not mean these
weights came from a Qwen3.5 model.1{
2 "algorithm": "official JANG_6M adaptive mixed-precision MSE quantization",
3 "bit_width": "approximately_6.2_average",
4 "group_size": "converter_auto",
5 "calibration_source": "none; official JANG MSE weight-only conversion"
6}JANG/MLX / 2.5.46.sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2 at
e8f8c211226b894fcb81acc59f3b34ba3efd5f42
as a measured proxy, not as ground-truth accuracy.1{
2 "release_gate": "PASS",
3 "text": [
4 true,
5 true,
6 true,
7 true,
8 true,
9 true,
10 true,
11 true,
12 true,
13 true
14 ],
15 "tools": [
16 true,
17 true,
18 true,
19 true,
20 true
21 ],
22 "vision": [
23 true,
24 true,
25 true
26 ],
27 "mtp": {
28 "passed": true,
29 "preserved": true,
30 "tensor_count": 31,
31 "metadata_mode": "preserved_enabled",
32 "runtime_supported": false,
33 "measurement_required": false,
34 "acceptance_rate": null,
35 "baseline_tps": null,
36 "mtp_tps": null,
37 "speedup": null,
38 "measured_improvement": false,
39 "advertise_acceleration": false,
40 "limitation": "JANG 2.5.46 preserves the Qwen3.8 MTP tensors, but its public mlx-vlm loader filters them because the model class has no native MTP module; no acceptance-rate or speed A/B can be measured."
41 },
42 "bf16_source_comparison": {
43 "passed": true,
44 "mean_semantic_similarity": 0.9883163690567016,
45 "exact_matches": 8,
46 "measurements": {
47 "average_generation_tps": 12.008561959684004,
48 "peak_memory_gb": 27.216688916,
49 "artifact_bytes": 23910017202,
50 "maximum_prompt_tokens_tested": 73,
51 "loop_rate": 0.0
52 },
53 "evaluator": {
54 "repo_id": "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
55 "revision": "e8f8c211226b894fcb81acc59f3b34ba3efd5f42",
56 "pooling": "attention-mask mean pooling followed by L2 normalization",
57 "maximum_tokens": 256
58 }
59 }
60}1python -m pip install vllm huggingface-hub
2vllm serve Chungulus/Qwen3.8-27B-JANG_6MJANG/MLX at or above 2.5.46; the exact validated container digest is recorded in quantization_manifest.json.enable_thinking,
reasoning_effort, and preserve_thinking) and the native Qwen tool format.validation_result.json; untested context lengths
must not be inferred from the architectural maximum.