Biography
I am a Ph.D. candidate in Computer Science at the University of Maryland, College Park, advised by Prof. Ang Li. My research has been recognized with the Qualcomm Innovation Fellowship (QIF) North America and the UMD CS Certificate of Outstanding Achievement. Currently, I am a Student Researcher at Google DeepMind & Ads working on efficient post-training. Previously, I conducted research at ByteDance Seed, Tencent AI Lab, and JD Explore Academy.
My research investigates the theoretical, algorithmic, and architectural principles governing efficient foundation models. In particular, I study how parameters, representational geometry, and computational paths interact inside large models—disentangling true model capacity from structural redundancy. Guided by these insights, I develop principled methods for conditional and dynamic computation (e.g., dynamic depth routing and expert load balancing / coordination in MoEs), representation-guided model compression (e.g., geometric pruning, layer dropping, and MoE compression), and parameter-efficient adaptation.
To ensure these algorithmic advances translate into practical real-world throughput, I design hardware- and system-aware paradigms for distributed serving and large-scale deployment (such as expert weight merging and dynamic sub-network routing). My recent work extends these principles toward unified multimodal sparsity, dense video understanding, and efficient dense retrieval across language and vision. Broadly, my goal is to bridge foundational model understanding, algorithm design, and system efficiency, enabling foundation models that are substantially faster, more economical, and universally accessible without compromising capability.
News
- [08/2026]: CoIn and Representation Evolution were accepted to EMNLP 2026.
- [08/2026]: Unified Multimodal Sparsity was accepted by TMLR.
- [08/2026]: Dense Video Understanding was accepted to ECCV 2026.
- [04/2026]: Demystifying Pruning and DualSparse-MoE were accepted to ICML 2026.
- [04/2026]: 📜 Received the Certificate of Outstanding Achievement from UMD CS.
- [04/2026]: EffiR was accepted to ACL 2026.
- [01/2026]: Capacity-Aware MoE was accepted to ICLR 2026.
- [01/2026]: Attention Drop was accepted by TMLR.
- [08/2025]: Router-Tuning was accepted to EMNLP 2025.
- [05/2025]: 🏆 Awarded the Qualcomm Innovation Fellowship (QIF) North America.
- [03/2025]: MoE Compression was accepted by TMLR.
Show earlier news (2022 – 2024)...
- [09/2024]: Two papers accepted: (1) Efficient Attention at NeurIPS 2024, and (2) Reformat Alignment at EMNLP 2024.
- [10/2023]: MEO was accepted to EMNLP 2023 (Oral).
- [05/2023]: PAD-Net was accepted to ACL 2023.
- [04/2023]: NeuralSlice was accepted to ICML 2023.
- [10/2022]: SparseAdapter was accepted to EMNLP 2022.
- [08/2022]: SD-Conv was accepted to WACV 2023.
- [07/2022]: 🏆 Ranked 1st in 4 tracks and top-3 in 7 tracks at WMT 2022.
- [01/2022]: Multi-modal Stock Prediction was accepted by AAAI-22 KDF.
Research Experience



Selected Publications
BibTeX
@article{he2026understanding,
title={Understanding and Harnessing Sparsity for Unified Multimodal Models},
author={He, Shwai and Deng, Chaorui and Li, Ang and Yan, Shen},
journal={Transactions on Machine Learning Research (TMLR)},
year={2026}
}BibTeX
@inproceedings{he2026demystifying,
title={Demystifying When Pruning Works via Representation Hierarchies},
author={He, Shwai and Sun, Guoheng and Zhang, Haichao and Fu, Yun and Li, Ang},
booktitle={Proceedings of the 43rd International Conference on Machine Learning (ICML)},
year={2026}
}BibTeX
@inproceedings{he2026capacity,
title={Capacity-Aware Inference: Mitigating the Straggler Effect in Mixture of Experts},
author={He, Shwai and Cai, Weilin and Huang, Jiayi and Li, Ang},
booktitle={Proceedings of the 14th International Conference on Learning Representations (ICLR)},
year={2026}
}BibTeX
@article{he2026uncovering,
title={Uncovering the Redundancy in Transformers via a Unified Study of Layer Dropping},
author={He, Shwai and Sun, Guoheng and Shen, Zheyu and Li, Ang},
journal={Transactions on Machine Learning Research (TMLR)},
year={2026}
}BibTeX
@inproceedings{lei2026making,
title={Making Large Language Models Efficient Dense Retrievers},
author={Lei, Yibin and He, Shwai and Li, Ang and Yates, Andrew},
booktitle={Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (ACL)},
year={2026}
}BibTeX
@inproceedings{he2025router,
title={Router-Tuning: A Simple and Effective Approach for Dynamic Depth},
author={He, Shwai and Ge, Tao and Sun, Guoheng and Tian, Bowei and Wang, Xiaoyang and Yu, Dong},
booktitle={Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing (EMNLP)},
year={2025}
}BibTeX
@article{he2025towards,
title={Towards Efficient Mixture of Experts: A Holistic Study of Compression Techniques},
author={He, Shwai and Dong, Daize and Ding, Liang and Li, Ang},
journal={Transactions on Machine Learning Research (TMLR)},
year={2025}
}BibTeX
@inproceedings{he2023merging,
title={Merging Experts into One: Improving Computational Efficiency of Mixture of Experts},
author={He, Shwai and Fan, Run-Ze and Ding, Liang and Shen, Li and Zhou, Tianyi and Tao, Dacheng},
booktitle={Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP)},
year={2023}
}BibTeX
@inproceedings{he2023pad,
title={PAD-Net: An Efficient Framework for Dynamic Networks},
author={He, Shwai and Ding, Liang and Dong, Daize and Liu, Boan and Yu, Fuqiang and Tao, Dacheng},
booktitle={Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (ACL)},
year={2023}
}BibTeX
@inproceedings{he2022sparseadapter,
title={SparseAdapter: An Easy Approach for Improving the Parameter-Efficiency of Adapters},
author={He, Shwai and Ding, Liang and Dong, Daize and Zhang, Miao and Tao, Dacheng},
booktitle={Findings of the Association for Computational Linguistics: EMNLP 2022},
year={2022}
}BibTeX
@inproceedings{he2023sdconv,
title={SD-Conv: Towards the Parameter-Efficiency of Dynamic Convolution},
author={He, Shwai and Jiang, Chenbo and Dong, Daize and Ding, Liang},
booktitle={IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)},
year={2023}
}BibTeX
@article{he2022multimodal,
title={Multi-modal Attention Network for Stock Movements Prediction},
author={He, Shwai and Gu, Shi},
journal={AAAI-22 Workshop on Knowledge Discovery from Unstructured Data in Financial Service},
year={2022}
}BibTeX
@inproceedings{cai2026dualsparse,
title={DualSparse-MoE: Coordinating Tensor/Neuron-Level Sparsity with Expert Partition and Reconstruction},
author={Cai, Weilin and Qin, Le and He, Shwai and Cui, Junwei and Li, Ang and Huang, Jiayi},
booktitle={Proceedings of the 43rd International Conference on Machine Learning (ICML)},
year={2026}
}BibTeX
@inproceedings{zhang2026dense,
title={Dense Video Understanding with Inter-tokenization Acceleration},
author={Zhang, Haichao and Chai, Wenhao and He, Shwai and Li, Ang and Fu, Yun},
booktitle={European Conference on Computer Vision (ECCV)},
year={2026}
}BibTeX
@inproceedings{sun2026coin,
title={CoIn: Counting the Invisible Reasoning Tokens in Commercial Opaque LLM APIs},
author={Sun, Guoheng and Wang, Ziyao and Tian, Bowei and Liu, Meng and Shen, Zheyu and He, Shwai and He, Yexiao and Ye, Wanghao and Wang, Yiting and Li, Ang},
booktitle={Proceedings of the 2026 Conference on Empirical Methods in Natural Language Processing (EMNLP)},
year={2026}
}BibTeX
@inproceedings{jiang2023neuralslice,
title={NeuralSlice: Neural 3D Triangle Mesh Reconstruction via Slicing 4D Tetrahedral Meshes},
author={Jiang, Chenbo and Yang, Jie and He, Shwai and Lai, Yu-Kun and Gao, Lin},
booktitle={Proceedings of the 40th International Conference on Machine Learning (ICML)},
year={2023}
}BibTeX
@inproceedings{zan2022vega,
title={Vega-MT: The JD Explore Academy Translation System for WMT},
author={Zan, Changtong and Peng, Keqin and Ding, Liang and Qiu, Baopu and Liu, Boan and He, Shwai and Lu, Qingyu and Zhang, Zheng and Liu, Chuang and Liu, Weifeng and Zhan, Yibing and Tao, Dacheng},
booktitle={Proceedings of the Seventh Conference on Machine Translation (WMT)},
year={2022}
}Teaching
- Spring 2025: Teaching Assistant for CMSC 320 (Introduction to Data Science)
- Fall 2024: Teaching Assistant for CMSC 250 (Discrete Structures)
- Spring 2024: Teaching Assistant for CMSC 351 (Algorithms)
Last updated: August 21, 2026
