Views
No views yet
1>>> from transformers import pipeline
2>>> question_answerer = pipeline("question-answering", model='distilbert-base-uncased-distilled-squad')
3
4>>> context = r"""
5... Extractive Question Answering is the task of extracting an answer from a text given a question. An example of a
6... question answering dataset is the SQuAD dataset, which is entirely based on that task. If you would like to fine-tune
7... a model on a SQuAD task, you may leverage the examples/pytorch/question-answering/run_squad.py script.
8... """
9
10>>> result = question_answerer(question="What is a good example of a question answering dataset?", context=context)
11>>> print(
12... f"Answer: '{result['answer']}', score: {round(result['score'], 4)}, start: {result['start']}, end: {result['end']}"
13...)
14
15Answer: 'SQuAD dataset', score: 0.4704, start: 147, end: 1601from transformers import DistilBertTokenizer, DistilBertForQuestionAnswering
2import torch
3tokenizer = DistilBertTokenizer.from_pretrained('distilbert-base-uncased-distilled-squad')
4model = DistilBertForQuestionAnswering.from_pretrained('distilbert-base-uncased-distilled-squad')
5
6question, text = "Who was Jim Henson?", "Jim Henson was a nice puppet"
7
8inputs = tokenizer(question, text, return_tensors="pt")
9with torch.no_grad():
10 outputs = model(**inputs)
11
12answer_start_index = torch.argmax(outputs.start_logits)
13answer_end_index = torch.argmax(outputs.end_logits)
14
15predict_answer_tokens = inputs.input_ids[0, answer_start_index : answer_end_index + 1]
16tokenizer.decode(predict_answer_tokens)1from transformers import DistilBertTokenizer, TFDistilBertForQuestionAnswering
2import tensorflow as tf
3
4tokenizer = DistilBertTokenizer.from_pretrained("distilbert-base-uncased-distilled-squad")
5model = TFDistilBertForQuestionAnswering.from_pretrained("distilbert-base-uncased-distilled-squad")
6
7question, text = "Who was Jim Henson?", "Jim Henson was a nice puppet"
8
9inputs = tokenizer(question, text, return_tensors="tf")
10outputs = model(**inputs)
11
12answer_start_index = int(tf.math.argmax(outputs.start_logits, axis=-1)[0])
13answer_end_index = int(tf.math.argmax(outputs.end_logits, axis=-1)[0])
14
15predict_answer_tokens = inputs.input_ids[0, answer_start_index : answer_end_index + 1]
16tokenizer.decode(predict_answer_tokens)1>>> from transformers import pipeline
2>>> question_answerer = pipeline("question-answering", model='distilbert-base-uncased-distilled-squad')
3
4>>> context = r"""
5... Alice is sitting on the bench. Bob is sitting next to her.
6... """
7
8>>> result = question_answerer(question="Who is the CEO?", context=context)
9>>> print(
10... f"Answer: '{result['answer']}', score: {round(result['score'], 4)}, start: {result['start']}, end: {result['end']}"
11...)
12
13Answer: 'Bob', score: 0.4183, start: 32, end: 35DistilBERT pretrained on the same data as BERT, which is BookCorpus, a dataset consisting of 11,038 unpublished books and English Wikipedia (excluding lists, tables and headers).
This model reaches a F1 score of 86.9 on the [SQuAD v1.1] dev set (for comparison, Bert bert-base-uncased version reaches a F1 score of 88.5).
1@inproceedings{sanh2019distilbert,
2 title={DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter},
3 author={Sanh, Victor and Debut, Lysandre and Chaumond, Julien and Wolf, Thomas},
4 booktitle={NeurIPS EMC^2 Workshop},
5 year={2019}
6}