Views
No views yet
1from transformers import RobertaForMaskedLM, RobertaTokenizerFast, RobertaModel
2
3# Load with MLM head
4model = RobertaForMaskedLM.from_pretrained("IlPakoZ/RNA-BERTa9700")
5tokenizer = RobertaTokenizerFast.from_pretrained("IlPakoZ/RNA-BERTa9700")
6
7# Alternatively, load only the encoder for downstream tasks
8encoder = RobertaModel.from_pretrained("IlPakoZ/RNA-BERTa9700")
9tokenizer = RobertaTokenizerFast.from_pretrained("IlPakoZ/RNA-BERTa9700")1@article{10.48550/arXiv.2203.15556,
2 title={Training compute-optimal large language models},
3 author={Hoffmann, Jordan and Borgeaud, Sebastian and Mensch, Arthur and Buchatskaya, Elena and Cai, Trevor and Rutherford, Eliza and Casas, Diego de Las and Hendricks, Lisa Anne and Welbl, Johannes and Clark, Aidan and others},
4 journal={arXiv preprint arXiv:2203.15556},
5 year={2022}
6},
7@article{10.1093/nar/gkaa921,
8 title={RNAcentral 2021: secondary structure integration, improved sequence search and new member databases},
9 journal={Nucleic acids research},
10 volume={49},
11 number={D1},
12 pages={D212--D220},
13 year={2021},
14 publisher={Oxford University Press}
15},
16@article{10.1093/nar/gkae979,
17 title={Database resources of the National Center for Biotechnology Information in 2025},
18 author={Sayers, Eric W and Beck, Jeffrey and Bolton, Evan E and Brister, J Rodney and Chan, Jessica and Connor, Ryan and Feldgarden, Michael and Fine, Anna M and Funk, Kathryn and Hoffman, Jinna and others},
19 journal={Nucleic acids research},
20 volume={53},
21 number={D1},
22 pages={D20--D29},
23 year={2025},
24 publisher={Oxford University Press}
25},
26@article{10.48550/arXiv.2203.03466,
27 title={Tensor programs v: Tuning large neural networks via zero-shot hyperparameter transfer},
28 author={Yang, Greg and Hu, Edward J and Babuschkin, Igor and Sidor, Szymon and Liu, Xiaodong and Farhi, David and Ryder, Nick and Pachocki, Jakub and Chen, Weizhu and Gao, Jianfeng},
29 journal={arXiv preprint arXiv:2203.03466},
30 year={2022}
31},
32
33@article{10.48550/arXiv.2112.11446,
34 title={Scaling language models: Methods, analysis \& insights from training gopher},
35 author={Rae, Jack W and Borgeaud, Sebastian and Cai, Trevor and Millican, Katie and Hoffmann, Jordan and Song, Francis and Aslanides, John and Henderson, Sarah and Ring, Roman and Young, Susannah and others},
36 journal={arXiv preprint arXiv:2112.11446},
37 year={2021}
38},
39
40@article{10.48550/arXiv.2407.17465,
41 title={u-$\mu$P: The Unit-Scaled Maximal Update Parametrization},
42 author={Blake, Charlie and Eichenberg, Constantin and Dean, Josef and Balles, Lukas and Prince, Luke Y and Deiseroth, Bj{\"o}rn and Cruz-Salinas, Andres Felipe and Luschi, Carlo and Weinbach, Samuel and Orr, Douglas},
43 journal={arXiv preprint arXiv:2407.17465},
44 year={2024}
45}
46