Views
No views yet
mdeberta-v3-base-subjectivity-sentiment-english) is primarily optimized for and evaluated on English news articles. While the broader research explored multilingual and zero-shot settings, performance on other languages might vary.| Training Loss | Epoch | Step | Validation Loss | Macro F1 | Macro P | Macro R | Subj F1 | Subj P | Subj R | Accuracy |
|---|---|---|---|---|---|---|---|---|---|---|
| No log | 1.0 | 52 | 0.7082 | 0.3383 | 0.7418 | 0.5062 | 0.0247 | 1.0 | 0.0125 | 0.4870 |
| No log | 2.0 | 104 | 0.6365 | 0.7639 | 0.7639 | 0.7643 | 0.7696 | 0.7811 | 0.7583 | 0.7641 |
| No log | 3.0 | 156 | 0.8187 | 0.6510 | 0.7128 | 0.6732 | 0.5822 | 0.8244 | 0.45 | 0.6645 |
| No log | 4.0 | 208 | 0.6188 | 0.7594 | 0.7653 | 0.7623 | 0.7506 | 0.8146 | 0.6958 | 0.7597 |
| No log | 5.0 | 260 | 0.6048 | 0.7617 | 0.7659 | 0.7641 | 0.7556 | 0.8095 | 0.7083 | 0.7619 |
| No log | 6.0 | 312 | 0.6011 | 0.7727 | 0.7749 | 0.7743 | 0.7702 | 0.8111 | 0.7333 | 0.7727 |
transformers pipeline for text classification:1import torch
2import torch.nn as nn
3from transformers import DebertaV2Model, DebertaV2Config, AutoTokenizer, PreTrainedModel, pipeline, AutoModelForSequenceClassification
4from transformers.models.deberta.modeling_deberta import ContextPooler
5
6sent_pipe = pipeline(
7 "sentiment-analysis",
8 model="cardiffnlp/twitter-xlm-roberta-base-sentiment",
9 tokenizer="cardiffnlp/twitter-xlm-roberta-base-sentiment",
10 top_k=None, # return all 3 sentiment scores
11)
12
13class CustomModel(PreTrainedModel):
14 config_class = DebertaV2Config
15 def __init__(self, config, sentiment_dim=3, num_labels=2, *args, **kwargs):
16 super().__init__(config, *args, **kwargs)
17 self.deberta = DebertaV2Model(config)
18 self.pooler = ContextPooler(config)
19 output_dim = self.pooler.output_dim
20 self.dropout = nn.Dropout(0.1)
21 self.classifier = nn.Linear(output_dim + sentiment_dim, num_labels)
22
23 def forward(self, input_ids, positive, neutral, negative, token_type_ids=None, attention_mask=None, labels=None):
24 outputs = self.deberta(input_ids=input_ids, attention_mask=attention_mask)
25 encoder_layer = outputs[0]
26 pooled_output = self.pooler(encoder_layer)
27 sentiment_features = torch.stack((positive, neutral, negative), dim=1).to(pooled_output.dtype)
28 combined_features = torch.cat((pooled_output, sentiment_features), dim=1)
29 logits = self.classifier(self.dropout(combined_features))
30 return {'logits': logits}
31
32model_name = "MatteoFasulo/mdeberta-v3-base-subjectivity-sentiment-english"
33tokenizer = AutoTokenizer.from_pretrained("microsoft/mdeberta-v3-base")
34config = DebertaV2Config.from_pretrained(
35 model_name,
36 num_labels=2,
37 id2label={0: 'OBJ', 1: 'SUBJ'},
38 label2id={'OBJ': 0, 'SUBJ': 1},
39 output_attentions=False,
40 output_hidden_states=False
41)
42model = CustomModel(config=config, sentiment_dim=3, num_labels=2).from_pretrained(model_name)
43
44def classify_subjectivity(text: str):
45 # get full sentiment distribution
46 dist = sent_pipe(text)[0]
47 pos = next(d["score"] for d in dist if d["label"] == "positive")
48 neu = next(d["score"] for d in dist if d["label"] == "neutral")
49 neg = next(d["score"] for d in dist if d["label"] == "negative")
50
51 # tokenize the text
52 inputs = tokenizer(text, padding=True, truncation=True, max_length=256, return_tensors='pt')
53
54 # feeding in the three sentiment scores
55 with torch.no_grad():
56 outputs = model(
57 input_ids=inputs["input_ids"],
58 attention_mask=inputs["attention_mask"],
59 positive=torch.tensor(pos).unsqueeze(0).float(),
60 neutral=torch.tensor(neu).unsqueeze(0).float(),
61 negative=torch.tensor(neg).unsqueeze(0).float()
62 )
63
64 # compute probabilities and pick the top label
65 probs = torch.softmax(outputs.get('logits')[0], dim=-1)
66 label = model.config.id2label[int(probs.argmax())]
67 score = probs.max().item()
68
69 return {"label": label, "score": score}
70
71examples = [
72 "The company reported a 10% increase in revenue for the last quarter.",
73 "And it could even be used to gather intelligence on Russian operations.",
74 "Dramatic pictures the next day show the charred and hollowed out relic of a once impressive and key Russian vessel.",
75 "Demands upon the public credit for social service are most difficult to resist."
76]
77for text in examples:
78 result = classify_subjectivity(text)
79 print(f"Text: {text}")
80 print(f"→ Subjectivity: {result['label']} (score={result['score']:.2f})\n")1@misc{fasulo2025aiwizardscheckthat2025,
2 title={AI Wizards at CheckThat! 2025: Enhancing Transformer-Based Embeddings with Sentiment for Subjectivity Detection in News Articles},
3 author={Matteo Fasulo and Luca Babboni and Luca Tedeschini},
4 year={2025},
5 eprint={2507.11764},
6 archivePrefix={arXiv},
7 primaryClass={cs.CL},
8 url={https://arxiv.org/abs/2507.11764},
9}