Views
No views yet
PJMixers-Dev/Gemma-3-Earthen-Completion-v0.1-4B was trained at 8K with batch size 4 gradient accumulation 2, so each step was 65,536 tokens (including any padding tokens). It was trained for 150 steps, adding up to a total of 9,830,400 unique tokens seen.1# Requirements before running
2# - Get latest commit of axolotl (currently c0a0c75)
3# - Download these to axolotl/src/axolotl/prompt_formatters
4# - https://github.com/xzuyn/axolotl/blob/came-plus-formatters/src/axolotl/prompt_strategies/formatter_regex.py
5# - https://github.com/xzuyn/axolotl/blob/came-plus-formatters/src/axolotl/prompt_strategies/customgemma3-regex.py
6# - pip install ftfy
7# - pip install git+https://github.com/xzuyn/CAME.git@sr-grams-cautious-8bit
8
9# Weights and Biases logging config
10wandb_project: Gemma-3-4B
11wandb_entity:
12wandb_watch:
13wandb_name: Gemma-3-Earthen-v0.1-4B-QLoRA-run1
14wandb_log_model:
15
16# Model checkpointing config
17output_dir: ./Outputs/Gemma-3-Earthen-v0.1-4B-QLoRA-run1
18save_steps: 10
19save_safetensors: true
20save_total_limit: 2
21save_only_model: true
22
23# Model architecture config
24base_model: PJMixers-Dev/Gemma-3-Earthen-Completion-v0.1-4B
25model_type: AutoModelForCausalLM
26tokenizer_type: AutoTokenizer
27
28# Mixed precision training config
29bf16: true
30fp16: false
31tf32: false
32
33# Model loading config
34load_in_8bit: false
35load_in_4bit: true
36strict: false
37
38# Sequence config
39sequence_len: 8192
40min_sample_len: 512
41sample_packing: true
42eval_sample_packing: true
43pad_to_sequence_len: true
44train_on_inputs: false
45group_by_length: false
46
47# LoRA adapter config
48adapter: qlora
49lora_model_dir:
50lora_r: 256
51lora_alpha: 256
52lora_dropout: 0.125
53lora_target_modules: 'language_model.model.layers.[\d]+.(mlp|cross_attn|self_attn).(up|down|gate|q|k|v|o)_proj'
54embeddings_skip_upcast: true
55
56# Dataset config
57datasets:
58# Gemma-3 Instruct
59 # Instruction Data
60 - path: PJMixers-Dev/allenai_tulu-3-sft-mixture-filtered-2-ShareGPT
61 type: customgemma3-regex
62 - path: OpenLeecher/lmsys_chat_1m_clean
63 type: customgemma3-regex
64 - path: PJMixers-Dev/Gryphe_Sonnet3.5-SlimOrcaDedupCleaned-20k-ShareGPT
65 type: customgemma3-regex
66 - path: jeffmeloy/sonnet3.5_science_conversations
67 type: customgemma3-regex
68 - path: PJMixers/grimulkan_theory-of-mind-ShareGPT
69 type: customgemma3-regex
70 - path: PJMixers/grimulkan_physical-reasoning-ShareGPT
71 type: customgemma3-regex
72 - path: TheDrummer/AmoralQA-v2
73 type: customgemma3-regex
74 - path: BeaverAI/REDACTED1
75 type: customgemma3-regex
76 - path: BeaverAI/REDACTED2
77 type: customgemma3-regex
78 - path: PJMixers/coedit-reworded-deduped-multiturn-sharegpt
79 type: customgemma3-regex
80 - path: Epiculous/Synthstruct-Gens-v1.1-Filtered-n-Cleaned
81 type: customgemma3-regex
82 - path: PJMixers/Anthropic_persuasion-ShareGPT
83 type: customgemma3-regex
84 - path: PJMixers/Math-Multiturn-1K-ShareGPT
85 type: customgemma3-regex
86 - path: PJMixers/AP-News-2024-CGPT-Summarize-ShareGPT
87 type: customgemma3-regex
88 - path: PJMixers-Dev/allenai_WildChat-1M-gemini-exp-1206-ShareGPT
89 type: customgemma3-regex
90 - path: PJMixers-Dev/allenai_WildChat-1M-gemini-2.0-flash-exp-ShareGPT
91 type: customgemma3-regex
92 - path: PJMixers-Dev/WizardLMTeam_WizardLM_evol_instruct_70k-gemini-2.0-flash-exp-ShareGPT
93 type: customgemma3-regex
94 - path: PocketDoc/Dans-MemoryCore-CoreCurriculum-Small
95 type: customgemma3-regex
96 - path: PocketDoc/Dans-Toolmaxx-Functions-ToolACE
97 type: customgemma3-regex
98 - path: PocketDoc/Dans-Codemaxx-LeetCode
99 type: customgemma3-regex
100 - path: PocketDoc/Dans-Codemaxx-CodeFeedback-SingleTurn
101 type: customgemma3-regex
102 # RP Data
103 - path: PJMixers-Dev/Gryphe-Aesir-RPG-Charcards-Opus-Mixed
104 type: customgemma3-regex
105 - path: allura-org/gryphe-sonnet-3.5-charcards-names-added
106 type: customgemma3-regex
107 - path: anthracite-org/c2_logs_32k_llama3_qwen2_v1.3
108 type: customgemma3-regex
109 - path: BeaverAI/REDACTED3
110 type: customgemma3-regex
111 - path: PJMixers-Dev/MinervaAI_Aesir-Preview-Anon
112 type: customgemma3-regex
113 - path: PJMixers-Dev/lemonilia_LimaRP-Simple-CustomShareGPT-Shuffled
114 type: customgemma3-regex
115 - path: Epiculous/SynthRP-Gens-v1.1-Filtered-n-Cleaned
116 type: customgemma3-regex
117 - path: PJMixers-Dev/NyxKrage_chub-logs-sharegpt-longest-CustomShareGPT
118 type: customgemma3-regex
119 - path: PJMixers/OpenLeecher_Teatime_all_logs_longest-ShareGPT
120 type: customgemma3-regex
121 - path: grimulkan/aicg-logs-augmented
122 type: customgemma3-regex
123 - path: grimulkan/PIPPA-augmented-dedup
124 type: customgemma3-regex
125 - path: PJMixers/grimulkan_bluemoon_Karen_cleaned-carded-formatted
126 type: customgemma3-regex
127 # InstStory Data
128 - path: PJMixers/lodrick-the-lafted_OpusStories-ShareGPT
129 type: customgemma3-regex
130 - path: Gryphe/ChatGPT-4o-Writing-Prompts
131 type: customgemma3-regex
132 - path: Gryphe/Opus-WritingPrompts
133 type: customgemma3-regex
134 - path: anthracite-org/nopm_claude_writing_fixed
135 type: customgemma3-regex
136 - path: PJMixers-Dev/Tiefighter-13B-Fake-Distill-ShareGPT
137 type: customgemma3-regex
138 - path: allura-org/fujin-instruct-v2
139 type: customgemma3-regex
140 # Adventure Data
141 - path: PocketDoc/Dans-Prosemaxx-Adventure
142 type: customgemma3-regex
143 - path: PocketDoc/Dans-Failuremaxx-Adventure-3
144 type: customgemma3-regex
145test_datasets:
146val_set_size: 128
147eval_strategy: steps
148eval_steps: 10
149dataset_prepared_path: ./00-Tokenized-Datasets/Gemma-3-Earthen-v0.1-4B-LoRA-seed42
150shuffle_merged_datasets: true
151dataset_processes:
152
153# Training hyperparameters
154num_epochs: 1
155gradient_accumulation_steps: 2
156micro_batch_size: 4
157eval_batch_size: 4
158warmup_steps: 0
159optimizer: came_pytorch
160optim_args:
161 enable_stochastic_rounding: true
162 enable_cautious: true
163 enable_8bit: true
164lr_scheduler: rex
165learning_rate: 5e-7
166cosine_min_lr_ratio: 0.05
167weight_decay: 0.01
168max_grad_norm: 0.5
169logging_steps: 1
170
171# Model optimization
172gradient_checkpointing: offload
173sdp_attention: true
174plugins:
175 - axolotl.integrations.liger.LigerPlugin
176 - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
177cut_cross_entropy: true
178liger_rope: true
179liger_rms_norm: true
180liger_layer_norm: true
181liger_glu_activation: true
182liger_cross_entropy: false
183liger_fused_linear_cross_entropy: false
184lora_mlp_kernel: false
185lora_qkv_kernel: false
186lora_o_kernel: false
187
188# DeepSpeed
189deepspeed:
190
191# Garbage Collection
192gc_steps:
193
194# Debug config
195debug: true
196seed: 42
197
198# Token config
199special_tokens:
200 bos_token: "<bos>"
201 eos_token: "<end_of_turn>"
202 pad_token: "<pad>"
203tokens:1@misc{wolf2020huggingfacestransformersstateoftheartnatural,
2 title={HuggingFace's Transformers: State-of-the-art Natural Language Processing},
3 author={Thomas Wolf and Lysandre Debut and Victor Sanh and Julien Chaumond and Clement Delangue and Anthony Moi and Pierric Cistac and Tim Rault and Rémi Louf and Morgan Funtowicz and Joe Davison and Sam Shleifer and Patrick von Platen and Clara Ma and Yacine Jernite and Julien Plu and Canwen Xu and Teven Le Scao and Sylvain Gugger and Mariama Drame and Quentin Lhoest and Alexander M. Rush},
4 year={2020},
5 eprint={1910.03771},
6 archivePrefix={arXiv},
7 primaryClass={cs.CL},
8 url={https://arxiv.org/abs/1910.03771},
9}
10@misc{gemmateam2025gemma3technicalreport,
11 title={Gemma 3 Technical Report},
12 author={Gemma Team and Aishwarya Kamath and Johan Ferret and Shreya Pathak and Nino Vieillard and Ramona Merhej and Sarah Perrin and Tatiana Matejovicova and Alexandre Ramé and Morgane Rivière and Louis Rouillard and Thomas Mesnard and Geoffrey Cideron and Jean-bastien Grill and Sabela Ramos and Edouard Yvinec and Michelle Casbon and Etienne Pot and Ivo Penchev and Gaël Liu and Francesco Visin and Kathleen Kenealy and Lucas Beyer and Xiaohai Zhai and Anton Tsitsulin and Robert Busa-Fekete and Alex Feng and Noveen Sachdeva and Benjamin Coleman and Yi Gao and Basil Mustafa and Iain Barr and Emilio Parisotto and David Tian and Matan Eyal and Colin Cherry and Jan-Thorsten Peter and Danila Sinopalnikov and Surya Bhupatiraju and Rishabh Agarwal and Mehran Kazemi and Dan Malkin and Ravin Kumar and David Vilar and Idan Brusilovsky and Jiaming Luo and Andreas Steiner and Abe Friesen and Abhanshu Sharma and Abheesht Sharma and Adi Mayrav Gilady and Adrian Goedeckemeyer and Alaa Saade and Alex Feng and Alexander Kolesnikov and Alexei Bendebury and Alvin Abdagic and Amit Vadi and András György and André Susano Pinto and Anil Das and Ankur Bapna and Antoine Miech and Antoine Yang and Antonia Paterson and Ashish Shenoy and Ayan Chakrabarti and Bilal Piot and Bo Wu and Bobak Shahriari and Bryce Petrini and Charlie Chen and Charline Le Lan and Christopher A. Choquette-Choo and CJ Carey and Cormac Brick and Daniel Deutsch and Danielle Eisenbud and Dee Cattle and Derek Cheng and Dimitris Paparas and Divyashree Shivakumar Sreepathihalli and Doug Reid and Dustin Tran and Dustin Zelle and Eric Noland and Erwin Huizenga and Eugene Kharitonov and Frederick Liu and Gagik Amirkhanyan and Glenn Cameron and Hadi Hashemi and Hanna Klimczak-Plucińska and Harman Singh and Harsh Mehta and Harshal Tushar Lehri and Hussein Hazimeh and Ian Ballantyne and Idan Szpektor and Ivan Nardini and Jean Pouget-Abadie and Jetha Chan and Joe Stanton and John Wieting and Jonathan Lai and Jordi Orbay and Joseph Fernandez and Josh Newlan and Ju-yeong Ji and Jyotinder Singh and Kat Black and Kathy Yu and Kevin Hui and Kiran Vodrahalli and Klaus Greff and Linhai Qiu and Marcella Valentine and Marina Coelho and Marvin Ritter and Matt Hoffman and Matthew Watson and Mayank Chaturvedi and Michael Moynihan and Min Ma and Nabila Babar and Natasha Noy and Nathan Byrd and Nick Roy and Nikola Momchev and Nilay Chauhan and Noveen Sachdeva and Oskar Bunyan and Pankil Botarda and Paul Caron and Paul Kishan Rubenstein and Phil Culliton and Philipp Schmid and Pier Giuseppe Sessa and Pingmei Xu and Piotr Stanczyk and Pouya Tafti and Rakesh Shivanna and Renjie Wu and Renke Pan and Reza Rokni and Rob Willoughby and Rohith Vallu and Ryan Mullins and Sammy Jerome and Sara Smoot and Sertan Girgin and Shariq Iqbal and Shashir Reddy and Shruti Sheth and Siim Põder and Sijal Bhatnagar and Sindhu Raghuram Panyam and Sivan Eiger and Susan Zhang and Tianqi Liu and Trevor Yacovone and Tyler Liechty and Uday Kalra and Utku Evci and Vedant Misra and Vincent Roseberry and Vlad Feinberg and Vlad Kolesnikov and Woohyun Han and Woosuk Kwon and Xi Chen and Yinlam Chow and Yuvein Zhu and Zichuan Wei and Zoltan Egyed and Victor Cotruta and Minh Giang and Phoebe Kirk and Anand Rao and Kat Black and Nabila Babar and Jessica Lo and Erica Moreira and Luiz Gustavo Martins and Omar Sanseviero and Lucas Gonzalez and Zach Gleicher and Tris Warkentin and Vahab Mirrokni and Evan Senter and Eli Collins and Joelle Barral and Zoubin Ghahramani and Raia Hadsell and Yossi Matias and D. Sculley and Slav Petrov and Noah Fiedel and Noam Shazeer and Oriol Vinyals and Jeff Dean and Demis Hassabis and Koray Kavukcuoglu and Clement Farabet and Elena Buchatskaya and Jean-Baptiste Alayrac and Rohan Anil and Dmitry and Lepikhin and Sebastian Borgeaud and Olivier Bachem and Armand Joulin and Alek Andreev and Cassidy Hardin and Robert Dadashi and Léonard Hussenot},
13 year={2025},
14 eprint={2503.19786},
15 archivePrefix={arXiv},
16 primaryClass={cs.CL},
17 url={https://arxiv.org/abs/2503.19786},
18}
19@misc{hu2021loralowrankadaptationlarge,
20 title={LoRA: Low-Rank Adaptation of Large Language Models},
21 author={Edward J. Hu and Yelong Shen and Phillip Wallis and Zeyuan Allen-Zhu and Yuanzhi Li and Shean Wang and Lu Wang and Weizhu Chen},
22 year={2021},
23 eprint={2106.09685},
24 archivePrefix={arXiv},
25 primaryClass={cs.CL},
26 url={https://arxiv.org/abs/2106.09685},
27}
28@misc{dettmers2023qloraefficientfinetuningquantized,
29 title={QLoRA: Efficient Finetuning of Quantized LLMs},
30 author={Tim Dettmers and Artidoro Pagnoni and Ari Holtzman and Luke Zettlemoyer},
31 year={2023},
32 eprint={2305.14314},
33 archivePrefix={arXiv},
34 primaryClass={cs.LG},
35 url={https://arxiv.org/abs/2305.14314},
36}
37@misc{dao2023flashattention2fasterattentionbetter,
38 title={FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning},
39 author={Tri Dao},
40 year={2023},
41 eprint={2307.08691},
42 archivePrefix={arXiv},
43 primaryClass={cs.LG},
44 url={https://arxiv.org/abs/2307.08691},
45}
46@misc{hsu2024ligerkernelefficienttriton,
47 title={Liger Kernel: Efficient Triton Kernels for LLM Training},
48 author={Pin-Lun Hsu and Yun Dai and Vignesh Kothapalli and Qingquan Song and Shao Tang and Siyu Zhu and Steven Shimizu and Shivam Sahni and Haowen Ning and Yanning Chen},
49 year={2024},
50 eprint={2410.10989},
51 archivePrefix={arXiv},
52 primaryClass={cs.LG},
53 url={https://arxiv.org/abs/2410.10989},
54}
55@misc{wijmans2025cutlosseslargevocabularylanguage,
56 title={Cut Your Losses in Large-Vocabulary Language Models},
57 author={Erik Wijmans and Brody Huval and Alexander Hertzberg and Vladlen Koltun and Philipp Krähenbühl},
58 year={2025},
59 eprint={2411.09009},
60 archivePrefix={arXiv},
61 primaryClass={cs.LG},
62 url={https://arxiv.org/abs/2411.09009},
63}
64@misc{chen2021rexrevisitingbudgetedtraining,
65 title={REX: Revisiting Budgeted Training with an Improved Schedule},
66 author={John Chen and Cameron Wolfe and Anastasios Kyrillidis},
67 year={2021},
68 eprint={2107.04197},
69 archivePrefix={arXiv},
70 primaryClass={cs.LG},
71 url={https://arxiv.org/abs/2107.04197},
72}
73@misc{luo2023cameconfidenceguidedadaptivememory,
74 title={CAME: Confidence-guided Adaptive Memory Efficient Optimization},
75 author={Yang Luo and Xiaozhe Ren and Zangwei Zheng and Zhuo Jiang and Xin Jiang and Yang You},
76 year={2023},
77 eprint={2307.02047},
78 archivePrefix={arXiv},
79 primaryClass={cs.CL},
80 url={https://arxiv.org/abs/2307.02047},
81}
82@misc{zamirai2021revisitingbfloat16training,
83 title={Revisiting BFloat16 Training},
84 author={Pedram Zamirai and Jian Zhang and Christopher R. Aberger and Christopher De Sa},
85 year={2021},
86 eprint={2010.06192},
87 archivePrefix={arXiv},
88 primaryClass={cs.LG},
89 url={https://arxiv.org/abs/2010.06192},
90}
91@misc{liang2025cautiousoptimizersimprovingtraining,
92 title={Cautious Optimizers: Improving Training with One Line of Code},
93 author={Kaizhao Liang and Lizhang Chen and Bo Liu and Qiang Liu},
94 year={2025},
95 eprint={2411.16085},
96 archivePrefix={arXiv},
97 primaryClass={cs.LG},
98 url={https://arxiv.org/abs/2411.16085},
99}
100@misc{xie2025sana15efficientscaling,
101 title={SANA 1.5: Efficient Scaling of Training-Time and Inference-Time Compute in Linear Diffusion Transformer},
102 author={Enze Xie and Junsong Chen and Yuyang Zhao and Jincheng Yu and Ligeng Zhu and Chengyue Wu and Yujun Lin and Zhekai Zhang and Muyang Li and Junyu Chen and Han Cai and Bingchen Liu and Daquan Zhou and Song Han},
103 year={2025},
104 eprint={2501.18427},
105 archivePrefix={arXiv},
106 primaryClass={cs.CV},
107 url={https://arxiv.org/abs/2501.18427},
108}
109@misc{dallabetta2024fundussimpletousenewsscraper,
110 title={Fundus: A Simple-to-Use News Scraper Optimized for High Quality Extractions},
111 author={Max Dallabetta and Conrad Dobberstein and Adrian Breiding and Alan Akbik},
112 year={2024},
113 eprint={2403.15279},
114 archivePrefix={arXiv},
115 primaryClass={cs.CL},
116 url={https://arxiv.org/abs/2403.15279},
117}
118@misc{chen2023meditron70bscalingmedicalpretraining,
119 title={MEDITRON-70B: Scaling Medical Pretraining for Large Language Models},
120 author={Zeming Chen and Alejandro Hernández Cano and Angelika Romanou and Antoine Bonnet and Kyle Matoba and Francesco Salvi and Matteo Pagliardini and Simin Fan and Andreas Köpf and Amirkeivan Mohtashami and Alexandre Sallinen and Alireza Sakhaeirad and Vinitra Swamy and Igor Krawczuk and Deniz Bayazit and Axel Marmet and Syrielle Montariol and Mary-Anne Hartley and Martin Jaggi and Antoine Bosselut},
121 year={2023},
122 eprint={2311.16079},
123 archivePrefix={arXiv},
124 primaryClass={cs.CL},
125 url={https://arxiv.org/abs/2311.16079},
126}
127@misc{lambert2025tulu3pushingfrontiers,
128 title={Tulu 3: Pushing Frontiers in Open Language Model Post-Training},
129 author={Nathan Lambert and Jacob Morrison and Valentina Pyatkin and Shengyi Huang and Hamish Ivison and Faeze Brahman and Lester James V. Miranda and Alisa Liu and Nouha Dziri and Shane Lyu and Yuling Gu and Saumya Malik and Victoria Graf and Jena D. Hwang and Jiangjiang Yang and Ronan Le Bras and Oyvind Tafjord and Chris Wilhelm and Luca Soldaini and Noah A. Smith and Yizhong Wang and Pradeep Dasigi and Hannaneh Hajishirzi},
130 year={2025},
131 eprint={2411.15124},
132 archivePrefix={arXiv},
133 primaryClass={cs.CL},
134 url={https://arxiv.org/abs/2411.15124},
135}
136@misc{gosling2023pippapartiallysyntheticconversational,
137 title={PIPPA: A Partially Synthetic Conversational Dataset},
138 author={Tear Gosling and Alpin Dale and Yinhe Zheng},
139 year={2023},
140 eprint={2308.05884},
141 archivePrefix={arXiv},
142 primaryClass={cs.CL},
143 url={https://arxiv.org/abs/2308.05884},
144}
145@misc{zheng2025opencodeinterpreterintegratingcodegeneration,
146 title={OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement},
147 author={Tianyu Zheng and Ge Zhang and Tianhao Shen and Xueling Liu and Bill Yuchen Lin and Jie Fu and Wenhu Chen and Xiang Yue},
148 year={2025},
149 eprint={2402.14658},
150 archivePrefix={arXiv},
151 primaryClass={cs.SE},
152 url={https://arxiv.org/abs/2402.14658},
153}
154@misc{liu2024toolacewinningpointsllm,
155 title={ToolACE: Winning the Points of LLM Function Calling},
156 author={Weiwen Liu and Xu Huang and Xingshan Zeng and Xinlong Hao and Shuai Yu and Dexun Li and Shuai Wang and Weinan Gan and Zhengying Liu and Yuanqing Yu and Zezhong Wang and Yuxian Wang and Wu Ning and Yutai Hou and Bin Wang and Chuhan Wu and Xinzhi Wang and Yong Liu and Yasheng Wang and Duyu Tang and Dandan Tu and Lifeng Shang and Xin Jiang and Ruiming Tang and Defu Lian and Qun Liu and Enhong Chen},
157 year={2024},
158 eprint={2409.00920},
159 archivePrefix={arXiv},
160 primaryClass={cs.LG},
161 url={https://arxiv.org/abs/2409.00920},
162}
163@misc{xia2025leetcodedatasettemporaldatasetrobust,
164 title={LeetCodeDataset: A Temporal Dataset for Robust Evaluation and Efficient Training of Code LLMs},
165 author={Yunhui Xia and Wei Shen and Yan Wang and Jason Klein Liu and Huifeng Sun and Siyue Wu and Jian Hu and Xiaolong Xu},
166 year={2025},
167 eprint={2504.14655},
168 archivePrefix={arXiv},
169 primaryClass={cs.LG},
170 url={https://arxiv.org/abs/2504.14655},
171}
172@misc{singh2024measuringimprovingpersuasivenesslarge,
173 title={Measuring and Improving Persuasiveness of Large Language Models},
174 author={Somesh Singh and Yaman K Singla and Harini SI and Balaji Krishnamurthy},
175 year={2024},
176 eprint={2410.02653},
177 archivePrefix={arXiv},
178 primaryClass={cs.CL},
179 url={https://arxiv.org/abs/2410.02653},
180}
181@misc{raheja2023coedittexteditingtaskspecific,
182 title={CoEdIT: Text Editing by Task-Specific Instruction Tuning},
183 author={Vipul Raheja and Dhruv Kumar and Ryan Koo and Dongyeop Kang},
184 year={2023},
185 eprint={2305.09857},
186 archivePrefix={arXiv},
187 primaryClass={cs.CL},
188 url={https://arxiv.org/abs/2305.09857},
189}
190@misc{zheng2024lmsyschat1mlargescalerealworldllm,
191 title={LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset},
192 author={Lianmin Zheng and Wei-Lin Chiang and Ying Sheng and Tianle Li and Siyuan Zhuang and Zhanghao Wu and Yonghao Zhuang and Zhuohan Li and Zi Lin and Eric P. Xing and Joseph E. Gonzalez and Ion Stoica and Hao Zhang},
193 year={2024},
194 eprint={2309.11998},
195 archivePrefix={arXiv},
196 primaryClass={cs.CL},
197 url={https://arxiv.org/abs/2309.11998},
198}