Based on WildChat.
Only kept gpt-4 completions with no toxicity and removed conversations that had empty content. Only English, Spanish, French, German and Italian samples are kept.
from datasets import load_dataset, DatasetDict
dataset = load_dataset("allenai/WildChat-1M", split="train")
unique_convos = set(dataset.unique("conversation_hash"))
LANGUAGES = ["english", "spanish", "french", "german", "italian"]
KEEP_MSG_KEYS = ["content", "role"]
avoid_words = ["gpt", "openai"]