Views
No views yet
| Rank | Model | Base Model | Method | Overall Score | Chat | Chat Hard | Safety | Reasoning |
|---|---|---|---|---|---|---|---|---|
| 1 | Decision-Tree-Reward-Gemma-2-27B | Gemma-2-27B | Decision Tree | 95.4 | 96.9 | 91.4 | 93.9 | 99.2 |
| 2 | INF-QRM-Llama3.1-70B | Llama-3.1-70B | Sequence Classifier | 95.1 | 96.6 | 91.0 | 93.6 | 99.1 |
| 3 | Decision-Tree-Reward-Llama-3.1-8B | Llama-3.1-8B | Decision Tree | 94.5 | 96.6 | 89.5 | 93.2 | 98.6 |
| 4 | QRM-Gemma-2-27B | Gemma-2-27B | Sequence Classifier | 94.4 | 96.6 | 90.1 | 92.7 | 98.3 |
| 5 | Skywork-Reward-Gemma-2-27B-v0.2 | Gemma-2-27B | Sequence Classifier | 94.3 | 96.1 | 89.9 | 93.0 | 98.1 |
| 6 | Llama-3.1-Nemotron-70B-Reward | Llama-3.1-70B | Custom Classifier | 94.1 | 97.5 | 85.7 | 95.1 | 98.1 |
| 7 | Skywork-Reward-Gemma-2-27B | Gemma-2-27B | Sequence Classifier | 93.8 | 95.8 | 91.4 | 91.9 | 96.1 |
| 8 | TextEval-Llama3.1-70B | Llama-3.1-70B | Generative | 93.5 | 94.1 | 90.1 | 93.2 | 96.4 |
| 9 | MetaMetrics-RM-v1.0 | - | Custom Classifier | 93.4 | 98.3 | 86.4 | 90.8 | 98.2 |
| 10 | Skywork-Critic-Llama-3.1-70B | Llama-3.1-70B | Generative | 93.3 | 96.6 | 87.9 | 93.1 | 95.5 |
| 11 | QRM-Llama3.1-8B-v2 | Llama-3.1-8B | Sequence Classifier | 93.1 | 96.4 | 86.8 | 92.6 | 96.8 |
| 12 | Skywork-Reward-Llama-3.1-8B-v0.2 | Llama-3.1-8B | Sequence Classifier | 93.1 | 94.7 | 88.4 | 92.7 | 96.7 |
transformers==4.45.2torch>=2.5.0flash-attn>=2.6.31from transformers import AutoModelForSequenceClassification
2import torch
3from transformers import AutoTokenizer
4model_name = "Decision-Tree-Reward-Llama-3.1-8B" # Another choice is "Decision-Tree-Reward-Gemma-2-27B"
5repo_id = f"RLHFlow/{model_name}"
6device = "cuda"
7# Initialize the model and tokenizer
8model = AutoModelForSequenceClassification.from_pretrained(repo_id, trust_remote_code=True, torch_dtype=torch.bfloat16, attn_implementation="flash_attention_2", device_map=device)
9tokenizer = AutoTokenizer.from_pretrained(repo_id, use_fast=True)
10# Load the decision tree
11model.load_decision_tree(repo_id, filename="decision_tree.pkl")
12
13# Prompt and response pairs
14prompt = "Jane has 12 apples. She gives 4 apples to her friend Mark, then buys 1 more apple, and finally splits all her apples equally among herself and her 2 siblings. How many apples does each person get?"
15response1 = "1. Jane starts with 12 apples and gives 4 to Mark. 12 - 4 = 8. Jane now has 8 apples.\n2. Jane buys 1 more apple. 8 + 1 = 9. Jane now has 9 apples.\n3. Jane splits the 9 apples equally among herself and her 2 siblings (3 people in total). 9 ÷ 3 = 3 apples each. Each person gets 3 apples."
16response2 = "1. Jane starts with 12 apples and gives 4 to Mark. 12 - 4 = 8. Jane now has 8 apples.\n2. Jane buys 1 more apple. 8 + 1 = 9. Jane now has 9 apples.\n3. Jane splits the 9 apples equally among her 2 siblings (2 people in total). 9 ÷ 2 = 4.5 apples each. Each person gets 4 apples."
17
18# Compare the two responses
19output = model.compare(prompt, response1, response2, tokenizer, device)
20print("Response 1 rewards")
21print(dict(zip(output["attributes"], output["rewards"][0])))
22# {'helpfulness': 3.9603815, 'correctness': 3.9727726, 'coherence': 3.8582935, 'complexity': 0.9909791, 'verbosity': 1.4901903}
23print("Response 2 rewards")
24print(dict(zip(output["attributes"], output["rewards"][1])))
25# {'helpfulness': 2.1698856, 'correctness': 2.2035594, 'coherence': 3.2032843, 'complexity': 0.8786768, 'verbosity': 1.4569137}
26print("Model preference")
27print(output["preference"])
28# 0
29