Views
No views yet
2026-01-30
1. Initial commit| File Size | Last Updated |
|---|---|
444 GiB | 2026-01-30 |
1from huggingface_hub import snapshot_download
2snapshot_download('QuantTrio/Kimi-K2.5-E304', cache_dir="your_local_path")
| Architecture | Mixture-of-Experts (MoE) |
| Total Parameters | 1T |
| Activated Parameters | 32B |
| Number of Layers (Dense layer included) | 61 |
| Number of Dense Layers | 1 |
| Attention Hidden Dimension | 7168 |
| MoE Hidden Dimension (per Expert) | 2048 |
| Number of Attention Heads | 64 |
| Number of Experts | 384 |
| Selected Experts per Token | 8 |
| Number of Shared Experts | 1 |
| Vocabulary Size | 160K |
| Context Length | 256K |
| Attention Mechanism | MLA |
| Activation Function | SwiGLU |
| Vision Encoder | MoonViT |
| Parameters of Vision Encoder | 400M |
| Benchmark | Kimi K2.5 (Thinking) | GPT-5.2 (xhigh) | Claude 4.5 Opus (Extended Thinking) | Gemini 3 Pro (High Thinking Level) | DeepSeek V3.2 (Thinking) | Qwen3-VL- 235B-A22B- Thinking |
|---|---|---|---|---|---|---|
| Reasoning & Knowledge | ||||||
| HLE-Full | 30.1 | 34.5 | 30.8 | 37.5 | 25.1† | - |
| HLE-Full (w/ tools) | 50.2 | 45.5 | 43.2 | 45.8 | 40.8† | - |
| AIME 2025 | 96.1 | 100 | 92.8 | 95.0 | 93.1 | - |
| HMMT 2025 (Feb) | 95.4 | 99.4 | 92.9* | 97.3* | 92.5 | - |
| IMO-AnswerBench | 81.8 | 86.3 | 78.5* | 83.1* | 78.3 | - |
| GPQA-Diamond | 87.6 | 92.4 | 87.0 | 91.9 | 82.4 | - |
| MMLU-Pro | 87.1 | 86.7* | 89.3* | 90.1 | 85.0 | - |
| Image & Video | ||||||
| MMMU-Pro | 78.5 | 79.5* | 74.0 | 81.0 | - | 69.3 |
| CharXiv (RQ) | 77.5 | 82.1 | 67.2* | 81.4 | - | 66.1 |
| MathVision | 84.2 | 83.0 | 77.1* | 86.1* | - | 74.6 |
| MathVista (mini) | 90.1 | 82.8* | 80.2* | 89.8* | - | 85.8 |
| ZeroBench | 9 | 9* | 3* | 8* | - | 4* |
| ZeroBench (w/ tools) | 11 | 7* | 9* | 12* | - | 3* |
| OCRBench | 92.3 | 80.7* | 86.5* | 90.3* | - | 87.5 |
| OmniDocBench 1.5 | 88.8 | 85.7 | 87.7* | 88.5 | - | 82.0* |
| InfoVQA (val) | 92.6 | 84* | 76.9* | 57.2* | - | 89.5 |
| SimpleVQA | 71.2 | 55.8* | 69.7* | 69.7* | - | 56.8* |
| WorldVQA | 46.3 | 28.0 | 36.8 | 47.4 | - | 23.5 |
| VideoMMMU | 86.6 | 85.9 | 84.4* | 87.6 | - | 80.0 |
| MMVU | 80.4 | 80.8* | 77.3 | 77.5 | - | 71.1 |
| MotionBench | 70.4 | 64.8 | 60.3 | 70.3 | - | - |
| VideoMME | 87.4 | 86.0* | - | 88.4* | - | 79.0 |
| LongVideoBench | 79.8 | 76.5* | 67.2* | 77.7* | - | 65.6* |
| LVBench | 75.9 | - | - | 73.5* | - | 63.6 |
| Coding | ||||||
| SWE-Bench Verified | 76.8 | 80.0 | 80.9 | 76.2 | 73.1 | - |
| SWE-Bench Pro | 50.7 | 55.6 | 55.4* | - | - | - |
| SWE-Bench Multilingual | 73.0 | 72.0 | 77.5 | 65.0 | 70.2 | - |
| Terminal Bench 2.0 | 50.8 | 54.0 | 59.3 | 54.2 | 46.4 | - |
| PaperBench | 63.5 | 63.7* | 72.9* | - | 47.1 | - |
| CyberGym | 41.3 | - | 50.6 | 39.9* | 17.3* | - |
| SciCode | 48.7 | 52.1 | 49.5 | 56.1 | 38.9 | - |
| OJBench (cpp) | 57.4 | - | 54.6* | 68.5* | 54.7* | - |
| LiveCodeBench (v6) | 85.0 | - | 82.2* | 87.4* | 83.3 | - |
| Long Context | ||||||
| Longbench v2 | 61.0 | 54.5* | 64.4* | 68.2* | 59.8* | - |
| AA-LCR | 70.0 | 72.3* | 71.3* | 65.3* | 64.3* | - |
| Agentic Search | ||||||
| BrowseComp | 60.6 | 65.8 | 37.0 | 37.8 | 51.4 | - |
| BrowseComp (w/ctx manage) | 74.9 | 57.8 | 59.2 | 67.6 | - | |
| BrowseComp (Agent Swarm) | 78.4 | - | - | - | - | - |
| WideSearch (iter-f1) | 72.7 | - | 76.2* | 57.0 | 32.5* | - |
| WideSearch (iter-f1 Agent Swarm) | 79.0 | - | - | - | - | - |
| DeepSearchQA | 77.1 | 71.3* | 76.1* | 63.2* | 60.9* | - |
| FinSearchCompT2&T3 | 67.8 | - | 66.2* | 49.9 | 59.1* | - |
| Seal-0 | 57.4 | 45.0 | 47.7* | 45.5* | 49.5* | - |
[!Note] You can access Kimi-K2.5's API on https://platform.moonshot.ai , we provide OpenAI/Anthropic-compatible API for you. To verify the deployment is correct, we also provide the Kimi Vendor Verifier. Currently, Kimi-K2.5 is recommended to run on the following inference engines:
[!Note]
Chat with video content is an experimental feature and is only supported in our official API for now The recommendedtemperaturewill be1.0for Thinking mode and0.6for Instant mode. The recommendedtop_pis0.95 To use instant mode, you need to pass{'chat_template_kwargs': {"thinking": False}}inextra_body.
1import openai
2import base64
3import requests
4def simple_chat(client: openai.OpenAI, model_name: str):
5 messages = [
6 {'role': 'system', 'content': 'You are Kimi, an AI assistant created by Moonshot AI.'},
7 {
8 'role': 'user',
9 'content': [
10 {'type': 'text', 'text': 'which one is bigger, 9.11 or 9.9? think carefully.'}
11 ],
12 },
13 ]
14 response = client.chat.completions.create(
15 model=model_name, messages=messages, stream=False, max_tokens=4096
16 )
17 print('===== Below is reasoning_content in Thinking Mode ======')
18 print(f'reasoning content: {response.choices[0].message.reasoning_content}')
19 print('===== Below is response in Thinking Mode ======')
20 print(f'response: {response.choices[0].message.content}')
21
22 # To use instant mode, pass {"thinking" = {"type":"disabled"}}
23 response = client.chat.completions.create(
24 model=model_name,
25 messages=messages,
26 stream=False,
27 max_tokens=4096,
28 extra_body={'thinking': {'type': 'disabled'}}, # this is for official API
29 # extra_body= {'chat_template_kwargs': {"thinking": False}} # this is for vLLM/SGLang
30 )
31 print('===== Below is response in Instant Mode ======')
32 print(f'response: {response.choices[0].message.content}')1import openai
2import base64
3import requests
4
5def chat_with_image(client: openai.OpenAI, model_name: str):
6 url = 'https://huggingface.co/moonshotai/Kimi-K2.5/resolve/main/figures/kimi-logo.png'
7 image_base64 = base64.b64encode(requests.get(url).content).decode()
8 messages = [
9 {
10 'role': 'user',
11 'content': [
12 {'type': 'text', 'text': 'Describe this image in detail.'},
13 {
14 'type': 'image_url',
15 'image_url': {'url': f'data:image/png;base64, {image_base64}'},
16 },
17 ],
18 }
19 ]
20
21 response = client.chat.completions.create(
22 model=model_name, messages=messages, stream=False, max_tokens=8192
23 )
24 print('===== Below is reasoning_content in Thinking Mode ======')
25 print(f'reasoning content: {response.choices[0].message.reasoning_content}')
26 print('===== Below is response in Thinking Mode ======')
27 print(f'response: {response.choices[0].message.content}')
28
29 # Also support instant mode if pass {"thinking" = {"type":"disabled"}}
30 response = client.chat.completions.create(
31 model=model_name,
32 messages=messages,
33 stream=False,
34 max_tokens=4096,
35 extra_body={'thinking': {'type': 'disabled'}}, # this is for official API
36 # extra_body= {'chat_template_kwargs': {"thinking": False}} # this is for vLLM/SGLang
37 )
38 print('===== Below is response in Instant Mode ======')
39 print(f'response: {response.choices[0].message.content}')
40
41 return response.choices[0].message.content1import openai
2import base64
3import requests
4
5def chat_with_video(client: openai.OpenAI, model_name:str):
6 url = 'https://huggingface.co/moonshotai/Kimi-K2.5/resolve/main/figures/demo_video.mp4'
7 video_base64 = base64.b64encode(requests.get(url).content).decode()
8 messages = [
9 {
10 "role": "user",
11 "content": [
12 {"type": "text","text": "Describe the video in detail."},
13 {
14 "type": "video_url",
15 "video_url": {"url": f"data:video/mp4;base64,{video_base64}"},
16 },
17 ],
18 }
19 ]
20
21 response = client.chat.completions.create(model=model_name, messages=messages)
22 print('===== Below is reasoning_content in Thinking Mode ======')
23 print(f'reasoning content: {response.choices[0].message.reasoning_content}')
24 print('===== Below is response in Thinking Mode ======')
25 print(f'response: {response.choices[0].message.content}')
26
27 # Also support instant mode if pass {"thinking" = {"type":"disabled"}}
28 response = client.chat.completions.create(
29 model=model_name,
30 messages=messages,
31 stream=False,
32 max_tokens=4096,
33 extra_body={'thinking': {'type': 'disabled'}}, # this is for official API
34 # extra_body= {'chat_template_kwargs': {"thinking": False}} # this is for vLLM/SGLang
35 )
36 print('===== Below is response in Instant Mode ======')
37 print(f'response: {response.choices[0].message.content}')
38 return response.choices[0].message.content