Views
No views yet
3.10 or greater, it is best that you install the version of PyTorch you would like to use, e.g. CPU/GPU version etc before installing this package else you will get the default version of PyTorch for your operating system/setup, but we do require torch>=2.2,<3.0.pip install wsd-torch-models1from transformers import AutoTokenizer
2import torch
3
4from wsd_torch_models.bem import BEM
5
6
7if __name__ == "__main__":
8 wsd_model_name = "ucrelnlp/PyMUSAS-Neural-English-Base-BEM"
9 wsd_model = BEM.from_pretrained(wsd_model_name)
10 tokenizer = AutoTokenizer.from_pretrained(wsd_model_name, add_prefix_space=True)
11
12 wsd_model.eval()
13 # Change this to the device you would like to use, e.g. cpu
14 model_device = "cpu"
15 wsd_model.to(device=model_device)
16
17 sentence = "The river bank was full of fish"
18 sentence_tokens = sentence.split()
19
20 with torch.inference_mode(mode=True):
21 # sub_word_tokenizer can be None when None it will download the appropriate tokenizer
22 # but generally it is better to give it the tokenizer as it saves the operation
23 # of checking if the tokenizer is already downloaded.
24 predictions = wsd_model.predict(sentence_tokens, sub_word_tokenizer=tokenizer, top_n=5)
25
26 for sentence_token, semantic_tags in zip(sentence_tokens, predictions):
27 print("Token: "+ sentence_token)
28 print("Most likely tags: ")
29 for tag in semantic_tags:
30 tag_definition = wsd_model.label_to_definition[tag]
31 print("\t" + tag + ":" + tag_definition)
32 print()| Parameter | 17M English | 68M English | 140M Multilingual | 307M Multilingual |
|---|---|---|---|---|
| Layers | 7 | 19 | 22 | 22 |
| Hidden Size | 256 | 512 | 384 | 768 |
| Intermediate Size | 384 | 768 | 1152 | 1152 |
| Attention Heads | 4 | 8 | 6 | 12 |
| Total Parameters | 17M | 68M | 140M | 307M |
| Non-embedding Parameters | 3.9M | 42.4M | 42M | 110M |
| Max Sequence Length | 8,000 | 8,000 | 8,192 | 8,192 |
| Vocabulary Size | 50,368 | 50,368 | 256,000 | 256,000 |
| Tokenizer | ModernBERT | ModernBERT | Gemma 2 | Gemma 2 |
| Dataset | 17M English | 68M English | 140M Multilingual | 307M Multilingual |
|---|---|---|---|---|
| Top 1 | ||||
| Chinese | - | - | 42.2 | 47.9 |
| English | 66.4 | 70.1 | 66.0 | 70.2 |
| Finnish | - | - | 15.8 | 25.9 |
| Irish | - | - | 28.5 | 35.6 |
| Welsh | - | - | 21.7 | 42.0 |
| Top 5 | ||||
| Chinese | - | - | 66.3 | 70.4 |
| English | 87.6 | 90.0 | 88.9 | 90.1 |
| Finnish | - | - | 32.8 | 42.4 |
| Irish | - | - | 47.6 | 51.6 |
| Welsh | - | - | 40.8 | 56.4 |
@misc{moore2026creatinghybridruleneural,
title={Creating a Hybrid Rule and Neural Network Based Semantic Tagger using Silver Standard Data: the PyMUSAS framework for Multilingual Semantic Annotation},
author={Andrew Moore and Paul Rayson and Dawn Archer and Tim Czerniak and Dawn Knight and Daisy Lal and Gearóid Ó Donnchadha and Mícheál Ó Meachair and Scott Piao and Elaine Uí Dhonnchadha and Johanna Vuorinen and Yan Yabo and Xiaobin Yang},
year={2026},
eprint={2601.09648},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2601.09648},
}