Biography
I am a Ph.D. candidate in Computer Science at the University of Maryland, College Park, advised by Prof. Ang Li. I study efficient language and multimodal models, connecting an understanding of model representations with methods for training, adaptation, and inference.
My work asks which computations a model needs, and how to allocate them. I investigate representation geometry and structural redundancy, then use these insights to develop dynamic depth routing, Mixture-of-Experts inference, and parameter-efficient adaptation. Recent projects extend this work to unified multimodal models, video understanding, and dense retrieval.
My research experience spans Google DeepMind & Ads (efficient post-training), ByteDance Seed (multimodal models), Tencent AI Lab, and JD Explore Academy. I received the Qualcomm Innovation Fellowship (QIF) North America and the UMD CS Certificate of Outstanding Achievement.
I am seeking full-time research opportunities in foundation models, inference acceleration, and efficient AI applications. Please get in touch to discuss potential opportunities.
News
- [08/2026]: Representation Evolution was accepted to EMNLP 2026.
- [08/2026]: Unified Multimodal Sparsity was accepted by TMLR.
- [08/2026]: Dense Video Understanding was accepted to ECCV 2026.
- [04/2026]: Demystifying Pruning and DualSparse-MoE were accepted to ICML 2026.
- [04/2026]: 📜 Received the Certificate of Outstanding Achievement from UMD CS.
- [04/2026]: EffiR was accepted to ACL 2026.
Earlier news (2022 – January 2026)
- [01/2026]: Capacity-Aware MoE was accepted to ICLR 2026.
- [01/2026]: Attention Drop was accepted by TMLR.
- [08/2025]: Router-Tuning was accepted to EMNLP 2025.
- [05/2025]: Awarded the Qualcomm Innovation Fellowship (QIF) North America.
- [03/2025]: MoE Compression was accepted by TMLR.
- [09/2024]: Two papers accepted: (1) Efficient Attention at NeurIPS 2024, and (2) Reformat Alignment at EMNLP 2024.
- [10/2023]: MEO was accepted to EMNLP 2023 (Oral).
- [05/2023]: PAD-Net was accepted to ACL 2023.
- [10/2022]: SparseAdapter was accepted to EMNLP 2022.
- [08/2022]: SD-Conv was accepted to WACV 2023.
- [07/2022]: 🏆 Ranked 1st in 4 tracks and top-3 in 7 tracks at WMT 2022.
- [01/2022]: Multi-modal Stock Prediction was accepted by AAAI-22 KDF.
Research Experience



Selected Publications
Model Understanding
BibTeX
@inproceedings{he2026disentangling,
title={Disentangling Representation Evolution in Transformers through Directional Decomposition},
author={He, Shwai and Zhang, Haichao and Yan, Shen},
booktitle={Proceedings of the 2026 Conference on Empirical Methods in Natural Language Processing (EMNLP)},
year={2026}
}BibTeX
@inproceedings{he2026demystifying,
title={Demystifying When Pruning Works via Representation Hierarchies},
author={He, Shwai and Sun, Guoheng and Zhang, Haichao and Fu, Yun and Li, Ang},
booktitle={Proceedings of the 43rd International Conference on Machine Learning (ICML)},
year={2026}
}BibTeX
@article{he2026uncovering,
title={Uncovering the Redundancy in Transformers via a Unified Study of Layer Dropping},
author={He, Shwai and Sun, Guoheng and Shen, Zheyu and Li, Ang},
journal={Transactions on Machine Learning Research (TMLR)},
year={2026}
}Inference Acceleration
BibTeX
@inproceedings{he2026capacity,
title={Capacity-Aware Inference: Mitigating the Straggler Effect in Mixture of Experts},
author={He, Shwai and Cai, Weilin and Huang, Jiayi and Li, Ang},
booktitle={Proceedings of the 14th International Conference on Learning Representations (ICLR)},
year={2026}
}BibTeX
@inproceedings{cai2026dualsparse,
title={DualSparse-MoE: Coordinating Tensor/Neuron-Level Sparsity with Expert Partition and Reconstruction},
author={Cai, Weilin and Qin, Le and He, Shwai and Cui, Junwei and Li, Ang and Huang, Jiayi},
booktitle={Proceedings of the 43rd International Conference on Machine Learning (ICML)},
year={2026}
}BibTeX
@inproceedings{he2025router,
title={Router-Tuning: A Simple and Effective Approach for Dynamic Depth},
author={He, Shwai and Ge, Tao and Sun, Guoheng and Tian, Bowei and Wang, Xiaoyang and Yu, Dong},
booktitle={Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing (EMNLP)},
year={2025}
}BibTeX
@article{he2025towards,
title={Towards Efficient Mixture of Experts: A Holistic Study of Compression Techniques},
author={He, Shwai and Dong, Daize and Ding, Liang and Li, Ang},
journal={Transactions on Machine Learning Research (TMLR)},
year={2025}
}BibTeX
@inproceedings{he2023merging,
title={Merging Experts into One: Improving Computational Efficiency of Mixture of Experts},
author={He, Shwai and Fan, Run-Ze and Ding, Liang and Shen, Li and Zhou, Tianyi and Tao, Dacheng},
booktitle={Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP)},
year={2023}
}BibTeX
@inproceedings{he2023pad,
title={PAD-Net: An Efficient Framework for Dynamic Networks},
author={He, Shwai and Ding, Liang and Dong, Daize and Liu, Boan and Yu, Fuqiang and Tao, Dacheng},
booktitle={Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (ACL)},
year={2023}
}BibTeX
@inproceedings{he2023sdconv,
title={SD-Conv: Towards the Parameter-Efficiency of Dynamic Convolution},
author={He, Shwai and Jiang, Chenbo and Dong, Daize and Ding, Liang},
booktitle={IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)},
year={2023}
}Efficient Adaptation
BibTeX
@inproceedings{lei2026making,
title={Making Large Language Models Efficient Dense Retrievers},
author={Lei, Yibin and He, Shwai and Li, Ang and Yates, Andrew},
booktitle={Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (ACL)},
year={2026}
}BibTeX
@inproceedings{he2022sparseadapter,
title={SparseAdapter: An Easy Approach for Improving the Parameter-Efficiency of Adapters},
author={He, Shwai and Ding, Liang and Dong, Daize and Zhang, Miao and Tao, Dacheng},
booktitle={Findings of the Association for Computational Linguistics: EMNLP 2022},
year={2022}
}Multimodal Learning & Vision
BibTeX
@article{he2026understanding,
title={Understanding and Harnessing Sparsity for Unified Multimodal Models},
author={He, Shwai and Deng, Chaorui and Li, Ang and Yan, Shen},
journal={Transactions on Machine Learning Research (TMLR)},
year={2026}
}BibTeX
@inproceedings{zhang2026dense,
title={Dense Video Understanding with Inter-tokenization Acceleration},
author={Zhang, Haichao and Chai, Wenhao and He, Shwai and Li, Ang and Fu, Yun},
booktitle={European Conference on Computer Vision (ECCV)},
year={2026}
}AI Applications
BibTeX
@article{jing2024accurate,
title={Accurate prediction of antibody function and structure using bio-inspired antibody language model},
author={Jing, Hongtai and Gao, Zhengtao and Xu, Sheng and Shen, Tao and Peng, Zhangzhi and He, Shwai and You, Tao and Ye, Shuang and Lin, Wei and Sun, Siqi},
journal={Briefings in Bioinformatics},
volume={25},
number={4},
pages={bbae245},
year={2024},
doi={10.1093/bib/bbae245}
}BibTeX
@article{he2022multimodal,
title={Multi-modal Attention Network for Stock Movements Prediction},
author={He, Shwai and Gu, Shi},
journal={AAAI-22 Workshop on Knowledge Discovery from Unstructured Data in Financial Service},
year={2022}
}Industry Model Reports
BibTeX
@misc{seed2026seed20,
title={Seed2.0 Model Card: Towards Intelligence Frontier for Real-World Complexity},
author={{ByteDance Seed}},
year={2026},
eprint={2607.00248},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2607.00248}
}BibTeX
@inproceedings{zan2022vega,
title={Vega-MT: The JD Explore Academy Translation System for WMT},
author={Zan, Changtong and Peng, Keqin and Ding, Liang and Qiu, Baopu and Liu, Boan and He, Shwai and Lu, Qingyu and Zhang, Zheng and Liu, Chuang and Liu, Weifeng and Zhan, Yibing and Tao, Dacheng},
booktitle={Proceedings of the Seventh Conference on Machine Translation (WMT)},
year={2022}
}Teaching
- Spring 2025: Teaching Assistant for CMSC 320 (Introduction to Data Science)
- Fall 2024: Teaching Assistant for CMSC 250 (Discrete Structures)
- Spring 2024: Teaching Assistant for CMSC 351 (Algorithms)
Last updated: September 21, 2026
