Views
No views yet
mDeBERTaV3-base, involves enhancing transformer-based embeddings by integrating sentiment scores derived from an auxiliary model. This sentiment-augmented architecture significantly boosts performance, especially the subjective F1 score. The model also employs decision threshold calibration, optimized on the development set, to effectively address class imbalance prevalent across the datasets.| Training Loss | Epoch | Step | Validation Loss | Macro F1 | Macro P | Macro R | Subj F1 | Subj P | Subj R | Accuracy |
|---|---|---|---|---|---|---|---|---|---|---|
| No log | 1.0 | 249 | 0.4619 | 0.7825 | 0.7790 | 0.7920 | 0.7421 | 0.6919 | 0.8 | 0.7901 |
| No log | 2.0 | 498 | 0.4544 | 0.7877 | 0.7856 | 0.8021 | 0.7537 | 0.6846 | 0.8384 | 0.7932 |
| 0.4852 | 3.0 | 747 | 0.5219 | 0.7931 | 0.7899 | 0.8052 | 0.7572 | 0.6970 | 0.8288 | 0.7994 |
| 0.4852 | 4.0 | 996 | 0.7055 | 0.7935 | 0.7975 | 0.7904 | 0.7385 | 0.7605 | 0.7178 | 0.8082 |
| 0.2503 | 5.0 | 1245 | 0.8536 | 0.7883 | 0.7937 | 0.7844 | 0.7306 | 0.7592 | 0.7041 | 0.8040 |
| 0.2503 | 6.0 | 1494 | 0.8900 | 0.7969 | 0.7988 | 0.7953 | 0.7450 | 0.7560 | 0.7342 | 0.8102 |
transformers library for text classification:1import torch
2import torch.nn as nn
3from transformers import DebertaV2Model, DebertaV2Config, AutoTokenizer, PreTrainedModel, pipeline, AutoModelForSequenceClassification
4from transformers.models.deberta.modeling_deberta import ContextPooler
5
6sent_pipe = pipeline(
7 "sentiment-analysis",
8 model="cardiffnlp/twitter-xlm-roberta-base-sentiment",
9 tokenizer="cardiffnlp/twitter-xlm-roberta-base-sentiment",
10 top_k=None, # return all 3 sentiment scores
11)
12
13class CustomModel(PreTrainedModel):
14 config_class = DebertaV2Config
15 def __init__(self, config, sentiment_dim=3, num_labels=2, *args, **kwargs):
16 super().__init__(config, *args, **kwargs)
17 self.deberta = DebertaV2Model(config)
18 self.pooler = ContextPooler(config)
19 output_dim = self.pooler.output_dim
20 self.dropout = nn.Dropout(0.1)
21 self.classifier = nn.Linear(output_dim + sentiment_dim, num_labels)
22
23 def forward(self, input_ids, positive, neutral, negative, token_type_ids=None, attention_mask=None, labels=None):
24 outputs = self.deberta(input_ids=input_ids, attention_mask=attention_mask)
25 encoder_layer = outputs[0]
26 pooled_output = self.pooler(encoder_layer)
27 sentiment_features = torch.stack((positive, neutral, negative), dim=1).to(pooled_output.dtype)
28 combined_features = torch.cat((pooled_output, sentiment_features), dim=1)
29 logits = self.classifier(self.dropout(combined_features))
30 return {'logits': logits}
31
32model_name = "MatteoFasulo/mdeberta-v3-base-subjectivity-sentiment-multilingual-no-arabic"
33tokenizer = AutoTokenizer.from_pretrained("microsoft/mdeberta-v3-base")
34config = DebertaV2Config.from_pretrained(
35 model_name,
36 num_labels=2,
37 id2label={0: 'OBJ', 1: 'SUBJ'},
38 label2id={'OBJ': 0, 'SUBJ': 1},
39 output_attentions=False,
40 output_hidden_states=False
41)
42model = CustomModel(config=config, sentiment_dim=3, num_labels=2).from_pretrained(model_name)
43
44def classify_subjectivity(text: str):
45 # get full sentiment distribution
46 dist = sent_pipe(text)[0]
47 pos = next(d["score"] for d in dist if d["label"] == "positive")
48 neu = next(d["score"] for d in dist if d["label"] == "neutral")
49 neg = next(d["score"] for d in dist if d["label"] == "negative")
50
51 # tokenize the text
52 inputs = tokenizer(text, padding=True, truncation=True, max_length=256, return_tensors='pt')
53
54 # feeding in the three sentiment scores
55 with torch.no_grad():
56 outputs = model(
57 input_ids=inputs["input_ids"],
58 attention_mask=inputs["attention_mask"],
59 positive=torch.tensor(pos).unsqueeze(0).float(),
60 neutral=torch.tensor(neu).unsqueeze(0).float(),
61 negative=torch.tensor(neg).unsqueeze(0).float()
62 )
63
64 # compute probabilities and pick the top label
65 probs = torch.softmax(outputs.get('logits')[0], dim=-1)
66 label = model.config.id2label[int(probs.argmax())]
67 score = probs.max().item()
68
69 return {"label": label, "score": score}
70
71examples = [
72 "The company reported a 10% increase in revenue for the last quarter.",
73 "Die angegebenen Fehlerquoten können daher nur für symptomatische Patienten gelten.",
74 "Si smonta qui definitivamente la narrazione per cui le scelte energetiche possono essere frutto esclusivo di valutazioni “tecniche” e non politiche.",
75]
76for text in examples:
77 result = classify_subjectivity(text)
78 print(f"Text: {text}")
79 print(f"→ Subjectivity: {result['label']} (score={result['score']:.2f})\n")
801@misc{fasulo2025aiwizardscheckthat2025,
2 title={AI Wizards at CheckThat! 2025: Enhancing Transformer-Based Embeddings with Sentiment for Subjectivity Detection in News Articles},
3 author={Matteo Fasulo and Luca Babboni and Luca Tedeschini},
4 year={2025},
5 eprint={2507.11764},
6 archivePrefix={arXiv},
7 primaryClass={cs.CL},
8 url={https://arxiv.org/abs/2507.11764},
9}