Views
No views yet
1import torch
2from transformers import AutoModelForTokenClassification, AutoTokenizer, AutoConfig
3
4tokenizer = AutoTokenizer.from_pretrained("fiveflow/roberta-base-spacing")
5roberta = AutoModelForTokenClassification.from_pretrained("fiveflow/roberta-base-spacing")
6
7org_text = "탄소중립과ESG경영에대한사회적요구확대".replace(" ", "") # 공백제거
8label = ["UNK", "PAD", "O", "B", "I", "E", "S"]
9# char 단위로 토큰화
10token_list = [tokenizer.cls_token_id]
11for char in org_text:
12 token_list.append(tokenizer.encode(char)[1])
13token_list.append(tokenizer.eos_token_id)
14tkd = torch.tensor(token_list).unsqueeze(0)
15
16output = roberta(tkd).logits
17
18_, pred_idx = torch.max(output, dim=2)
19tags = [label[idx] for idx in pred_idx.squeeze()][1:-1]
20pred_sent = ""
21for char_idx, spc_idx in enumerate(pred_idx.squeeze()[1:-1]):
22 # "E" tag 단위로 띄어쓰기
23 if label[spc_idx] == "E": pred_sent += org_text[char_idx] + " "
24 else: pred_sent += org_text[char_idx]
25
26print(pred_sent.strip())
27# '탄소중립과 ESG 경영에 대한 사회적 요구 확대'1@misc{park2021klue,
2 title={KLUE: Korean Language Understanding Evaluation},
3 author={Sungjoon Park and Jihyung Moon and Sungdong Kim and Won Ik Cho and Jiyoon Han and Jangwon Park and Chisung Song and Junseong Kim and Yongsook Song and Taehwan Oh and Joohong Lee and Juhyun Oh and Sungwon Lyu and Younghoon Jeong and Inkwon Lee and Sangwoo Seo and Dongjun Lee and Hyunwoo Kim and Myeonghwa Lee and Seongbo Jang and Seungwon Do and Sunkyoung Kim and Kyungtae Lim and Jongwon Lee and Kyumin Park and Jamin Shin and Seonghyun Kim and Lucy Park and Alice Oh and Jungwoo Ha and Kyunghyun Cho},
4 year={2021},
5 eprint={2105.09680},
6 archivePrefix={arXiv},
7 primaryClass={cs.CL}
8}
9