Views
No views yet
Note
All the pretrained models are uploaded in Huggingface Model Hub. Check https://huggingface.co/BM-K
1import torch
2from transformers import AutoModel, AutoTokenizer
3
4def cal_score(a, b):
5 if len(a.shape) == 1: a = a.unsqueeze(0)
6 if len(b.shape) == 1: b = b.unsqueeze(0)
7
8 a_norm = a / a.norm(dim=1)[:, None]
9 b_norm = b / b.norm(dim=1)[:, None]
10 return torch.mm(a_norm, b_norm.transpose(0, 1)) * 100
11
12model = AutoModel.from_pretrained('BM-K/KoSimCSE-roberta-multitask') # or 'BM-K/KoSimCSE-bert-multitask'
13tokenizer = AutoTokenizer.from_pretrained('BM-K/KoSimCSE-roberta-multitask') # or 'BM-K/KoSimCSE-bert-multitask'
14
15sentences = ['치타가 들판을 가로 질러 먹이를 쫓는다.',
16 '치타 한 마리가 먹이 뒤에서 달리고 있다.',
17 '원숭이 한 마리가 드럼을 연주한다.']
18
19inputs = tokenizer(sentences, padding=True, truncation=True, return_tensors="pt")
20embeddings, _ = model(**inputs, return_dict=False)
21
22score01 = cal_score(embeddings[0][0], embeddings[1][0]) # 84.09
23# '치타가 들판을 가로 질러 먹이를 쫓는다.' @ '치타 한 마리가 먹이 뒤에서 달리고 있다.'
24score02 = cal_score(embeddings[0][0], embeddings[2][0]) # 23.21
25# '치타가 들판을 가로 질러 먹이를 쫓는다.' @ '원숭이 한 마리가 드럼을 연주한다.'| Model | Embedding size | Hidden size | # Layers | # Heads |
|---|---|---|---|---|
| KLUE-BERT-base | 768 | 768 | 12 | 12 |
| KLUE-RoBERTa-base | 768 | 768 | 12 | 12 |
Warning
Large pre-trained models need a lot of GPU memory to train
| Model | Average | Cosine Pearson | Cosine Spearman | Euclidean Pearson | Euclidean Spearman | Manhattan Pearson | Manhattan Spearman | Dot Pearson | Dot Spearman |
|---|---|---|---|---|---|---|---|---|---|
| KoSBERT†SKT | 77.40 | 78.81 | 78.47 | 77.68 | 77.78 | 77.71 | 77.83 | 75.75 | 75.22 |
| KoSBERT | 80.39 | 82.13 | 82.25 | 80.67 | 80.75 | 80.69 | 80.78 | 77.96 | 77.90 |
| KoSRoBERTa | 81.64 | 81.20 | 82.20 | 81.79 | 82.34 | 81.59 | 82.20 | 80.62 | 81.25 |
| KoSentenceBART | 77.14 | 79.71 | 78.74 | 78.42 | 78.02 | 78.40 | 78.00 | 74.24 | 72.15 |
| KoSentenceT5 | 77.83 | 80.87 | 79.74 | 80.24 | 79.36 | 80.19 | 79.27 | 72.81 | 70.17 |
| KoSimCSE-BERT†SKT | 81.32 | 82.12 | 82.56 | 81.84 | 81.63 | 81.99 | 81.74 | 79.55 | 79.19 |
| KoSimCSE-BERT | 83.37 | 83.22 | 83.58 | 83.24 | 83.60 | 83.15 | 83.54 | 83.13 | 83.49 |
| KoSimCSE-RoBERTa | 83.65 | 83.60 | 83.77 | 83.54 | 83.76 | 83.55 | 83.77 | 83.55 | 83.64 |
| KoSimCSE-BERT-multitask | 85.71 | 85.29 | 86.02 | 85.63 | 86.01 | 85.57 | 85.97 | 85.26 | 85.93 |
| KoSimCSE-RoBERTa-multitask | 85.77 | 85.08 | 86.12 | 85.84 | 86.12 | 85.83 | 86.12 | 85.03 | 85.99 |
| Model | Average | Cosine Pearson | Cosine Spearman | Euclidean Pearson | Euclidean Spearman | Manhattan Pearson | Manhattan Spearman | Dot Pearson | Dot Spearman |
|---|---|---|---|---|---|---|---|---|---|
| KoSRoBERTa-base† | N/A | N/A | 48.96 | N/A | N/A | N/A | N/A | N/A | N/A |
| KoSRoBERTa-large† | N/A | N/A | 51.35 | N/A | N/A | N/A | N/A | N/A | N/A |
| KoSimCSE-BERT | 74.08 | 74.92 | 73.98 | 74.15 | 74.22 | 74.07 | 74.07 | 74.15 | 73.14 |
| KoSimCSE-RoBERTa | 75.27 | 75.93 | 75.00 | 75.28 | 75.01 | 75.17 | 74.83 | 75.95 | 75.01 |
| KoDiffCSE-RoBERTa | 77.17 | 77.73 | 76.96 | 77.21 | 76.89 | 77.11 | 76.81 | 77.74 | 76.97 |
1@misc{park2021klue,
2 title={KLUE: Korean Language Understanding Evaluation},
3 author={Sungjoon Park and Jihyung Moon and Sungdong Kim and Won Ik Cho and Jiyoon Han and Jangwon Park and Chisung Song and Junseong Kim and Yongsook Song and Taehwan Oh and Joohong Lee and Juhyun Oh and Sungwon Lyu and Younghoon Jeong and Inkwon Lee and Sangwoo Seo and Dongjun Lee and Hyunwoo Kim and Myeonghwa Lee and Seongbo Jang and Seungwon Do and Sunkyoung Kim and Kyungtae Lim and Jongwon Lee and Kyumin Park and Jamin Shin and Seonghyun Kim and Lucy Park and Alice Oh and Jung-Woo Ha and Kyunghyun Cho},
4 year={2021},
5 eprint={2105.09680},
6 archivePrefix={arXiv},
7 primaryClass={cs.CL}
8}
9
10@inproceedings{gao2021simcse,
11 title={{SimCSE}: Simple Contrastive Learning of Sentence Embeddings},
12 author={Gao, Tianyu and Yao, Xingcheng and Chen, Danqi},
13 booktitle={Empirical Methods in Natural Language Processing (EMNLP)},
14 year={2021}
15}
16
17@article{ham2020kornli,
18 title={KorNLI and KorSTS: New Benchmark Datasets for Korean Natural Language Understanding},
19 author={Ham, Jiyeon and Choe, Yo Joong and Park, Kyubyong and Choi, Ilji and Soh, Hyungjoon},
20 journal={arXiv preprint arXiv:2004.03289},
21 year={2020}
22}
23
24@inproceedings{reimers-2019-sentence-bert,
25 title = "Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks",
26 author = "Reimers, Nils and Gurevych, Iryna",
27 booktitle = "Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing",
28 month = "11",
29 year = "2019",
30 publisher = "Association for Computational Linguistics",
31 url = "http://arxiv.org/abs/1908.10084",
32}
33
34@inproceedings{chuang2022diffcse,
35 title={{DiffCSE}: Difference-based Contrastive Learning for Sentence Embeddings},
36 author={Chuang, Yung-Sung and Dangovski, Rumen and Luo, Hongyin and Zhang, Yang and Chang, Shiyu and Soljacic, Marin and Li, Shang-Wen and Yih, Wen-tau and Kim, Yoon and Glass, James},
37 booktitle={Annual Conference of the North American Chapter of the Association for Computational Linguistics (NAACL)},
38 year={2022}
39}