Views
No views yet
conda create -n gesturelsm python=3.12
conda activate gesturelsm
conda install pytorch==2.1.2 torchvision==0.16.2 torchaudio==2.1.2 pytorch-cuda=11.8 -c pytorch -c nvidia
pip install -r requirements.txt
bash demo/install_mfa.sh# Download the pretrained model (Shortcut) + (Shortcut-reflow) + (Diffusion) + (RVQ-VAEs)
gdown https://drive.google.com/drive/folders/1OfYWWJbaXal6q7LttQlYKWAy0KTwkPRw?usp=drive_link -O ./ckpt --folder
# Download the SMPL model
gdown https://drive.google.com/drive/folders/1MCks7CMNBtAzU2XihYezNmiGT_6pWex8?usp=drive_link -O ./datasets/hub --folderFor evaluation and training, not necessary for running a web demo or inference.
bash preprocess/bash_raw_cospeech_download.shRequire download dataset
# Evaluate the pretrained shortcut model (20 steps)
python test.py -c configs/shortcut_rvqvae_128.yaml
# Evaluate the pretrained shortcut-reflow model (2-step)
python test.py -c configs/shortcut_reflow_test.yaml
# Evaluate the pretrained diffusion model
python test.py -c configs/diffuser_rvqvae_128.yaml
Require download dataset
bash train_rvq.shRequire download dataset
# Train the shortcut model
python train.py -c configs/shortcut_rvqvae_128.yaml
# Train the diffusion model
python train.py -c configs/diffuser_rvqvae_128.yamlpython demo.py -c configs/shortcut_rvqvae_128_hf.yaml1@misc{liu2025gesturelsmlatentshortcutbased,
2 title={GestureLSM: Latent Shortcut based Co-Speech Gesture Generation with Spatial-Temporal Modeling},
3 author={Pinxin Liu and Luchuan Song and Junhua Huang and Chenliang Xu},
4 year={2025},
5 eprint={2501.18898},
6 archivePrefix={arXiv},
7 primaryClass={cs.CV},
8 url={https://arxiv.org/abs/2501.18898},
9}