Views
No views yet
1from transformers import AutoTokenizer
2
3hi_tokenizer = AutoTokenizer.from_pretrained('krinal/BertWordPieceTokenizer-hi')
4
5hi_str = "आज का सूर्य देखो, कितना प्यारा, कितना शीतल है"
6
7# encode text
8encoded_str = hi_tokenizer.encode(hi_str)
9
10# decode text
11decoded_str = hi_tokenizer.decode(encoded_str)1@article{kumar2019bhaav,
2 title={BHAAV-A Text Corpus for Emotion Analysis from Hindi Stories},
3 author={Kumar, Yaman and Mahata, Debanjan and Aggarwal, Sagar and Chugh, Anmol and Maheshwari, Rajat and Shah, Rajiv Ratn},
4 journal={arXiv preprint arXiv:1910.04073},
5 year={2019}
6}