I am a Researcher at RBC Borealis. I recently received my Ph.D. from University of Waterloo, where I was supervised by Prof. Krzysztof Czarnecki.
I obtained my master’s degree under the supervision of Prof. Lijun Zhang in the LAMDA Group led by Prof. Zhihua Zhou at Nanjing University.
I have also had research experience at Amazon, SONY, Borealis, Tsinghua University with Prof. Jingjing Liu and Prof. Yang Liu, Tencent Lightspeed & Quantum Studios, Alibaba, and Netease Games.
My research focuses on reliable and efficient multimodal learning, with a focus on vision-language models and multimodal retrieval.
I am always happy to connect about research opportunities and collaborations in multimodal learning (reliability, efficiency, and retrieval).
Yimu Wang
Researcher, RBC Borealis
Ph.D., University of Waterloo
@article{wang2026does,
title={Where Does the Answer Come From? Benchmarking View-Level Visual Evidence Identification in Multi-View MLLMs for Autonomous Driving},
author={Wang, Yimu and Choi, Yee Man and Zhang, Barry and Azadani, Mozhgan Nasr and Sedwards, Sean and Czarnecki, Krzysztof},
journal={arXiv preprint arXiv:2606.09644},
year={2026}
}
VISTAQA: Benchmarking Joint Visual Question Answering and Pixel-Level Evidence
Mozhgan Nasr Azadani, Yimu Wang, Yongpeng Zhu, Lihong Chen, Milan Ganai, Sean Sedwards, Marco Pavone, Krzysztof Czarnecki arXiv preprint, 2026.
@article{nasr2026vistaqa,
title={VISTAQA: Benchmarking Joint Visual Question Answering and Pixel-Level Evidence},
author={Nasr Azadani, Mozhgan and Wang, Yimu and Zhu, Yongpeng and Chen, Lihong and Ganai, Milan and Sedwards, Sean and Pavone, Marco and Czarnecki, Krzysztof},
journal={arXiv e-prints},
pages={arXiv--2605},
year={2026}
}
UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models
@inproceedings{wang2026uniform,
title={UNIFORM: Unifying Knowledge from Large-scale and Diverse Pre-trained Models},
author={Wang, Yimu and Zhuang, Weiming and Chen, Chen and Huang, Jiabo and Li, Jingtao and Lyu, Lingjuan},
booktitle={CVPR Findings},
year={2026}
}
Mitigating the Modality Gap: Few-Shot Out-of-Distribution Detection with Multi-modal Prototypes and Image Bias Estimation
Yimu Wang, Evelien Riddell, Adrian Chow, Sean Sedwards, Krzysztof Czarnecki IEEE Winter Conference on Applications of Computer Vision (WACV), 2026.
@inproceedings{wang2026mitigating,
title={Mitigating the modality gap: Few-shot out-of-distribution detection with multi-modal prototypes and image bias estimation},
author={Wang, Yimu and Riddell, Evelien and Chow, Adrian and Sedwards, Sean and Czarnecki, Krzysztof},
booktitle={WACV},
year={2026}
}
Lexicographic Lipschitz Bandits: New Algorithms and a Lower Bound
Bo Xue, Ji Cheng, Fei Liu, Yimu Wang, Lijun Zhang, and Qingfu Zhang Journal of Machine Learning Research (JMLR), 2025.
@article{xue2025lexicographic,
title={Lexicographic Lipschitz Bandits: New Algorithms and a Lower Bound},
author={Xue, Bo and Cheng, Ji and Liu, Fei and Wang, Yimu and Zhang, Lijun and Zhang, Qingfu},
journal={Journal of Machine Learning Research},
volume={26},
number={223},
pages={1--56},
year={2025}
}
Hawaii: Hierarchical Visual Knowledge Transfer for Efficient Vision-Language Models
Yimu Wang, Mozhgan Nasr Azadani, Sean Sedwards, Krzysztof Czarnecki Annual Conference on Neural Information Processing Systems (NeurIPS), 2025.
@inproceedings{wang2025hawaii,
title={HAWAII: Hierarchical Visual Knowledge Transfer for Efficient Vision-Language Models},
author={Wang, Yimu and Azadani, Mozhgan Nasr and Sedwards, Sean and Czarnecki, Krzysztof},
booktitle={NeurIPS 2025},
year={2025}
}
Survey of Video Diffusion Models: Foundations, Implementations, and Applications
Yimu Wang, Xuye Liu, Wei Pang, Li Ma, Shuai Yuan, Paul Debevec, Ning Yu Transactions on Machine Learning Research (TMLR), 2025.
@article{
wang2025survey,
title={Survey of Video Diffusion Models: Foundations, Implementations, and Applications},
author={Yimu Wang and Xuye Liu and Wei Pang and Li Ma and Shuai Yuan and Paul Debevec and Ning Yu},
journal={Transactions on Machine Learning Research},
issn={2835-8856},
year={2025},
url={https://openreview.net/forum?id=2ODDBObKjH},
note={Survey Certification}
}
LEO-MINI: An Efficient Multimodal Large Language Model using Conditional Token Reduction and Mixture of Multi-Modal Experts
Yimu Wang, Mozhgan Nasr Azadani, Sean Sedwards, Krzysztof Czarnecki Empirical Methods in Natural Language Processing (EMNLP), 2025.
@inproceedings{wang2025leo,
title={LEO-MINI: An Efficient Multimodal Large Language Model using Conditional Token Reduction and Mixture of Multi-Modal Experts},
author={Wang, Yimu and Azadani, Mozhgan Nasr and Sedwards, Sean and Czarnecki, Krzysztof},
booktitle={EMNLP 2025},
year={2025}
}
Rethinking Spectral Augmentation for Contrast-based Graph Self-Supervised Learning
@article{jian2025rethinking,
title={Rethinking Spectral Augmentation for Contrast-based Graph Self-Supervised Learning},
author={Jian, Xiangru and Zhao, Xinjian and Pang, Wei and Ying, Chaolong and Wang, Yimu and Xu, Yaoyao and Yu, Tianshu},
journal={Transactions on Machine Learning Research},
year={2025}
}
OV-SCAN: Semantically Consistent Alignment for Novel Object Discovery in Open-Vocabulary 3D Object Detection
A. Chow, E. Riddell,Yimu Wang, S. Sedwards, K. Czarnecki International Conference on Computer Vision (ICCV), 2025.
@inproceedings{chow2025ov,
title={OV-SCAN: Semantically Consistent Alignment for Novel Object Discovery in Open-Vocabulary 3D Object Detection},
author={Chow, Adrian and Riddell, Evelien and Wang, Yimu and Sedwards, Sean and Czarnecki, Krzysztof},
booktitle={ICCV},
year={2025}
}
NBDESCRIB: A Dataset for Text Description Generation from Tables and Code in Jupyter Notebooks with Guidelines
Xuye Liu, Tengfei Ma,Yimu Wang, Fengjie Wang, Jian Zhao Annual Meeting of the Association for Computational Linguistics (Findings of ACL), 2025.
Findings of ACL 2025
@inproceedings{liu2025nbdescrib,
title={NBDESCRIB: A Dataset for Text Description Generation from Tables and Code in Jupyter Notebooks with Guidelines},
author={Liu, Xuye and Ma, Tengfei and Wang, Yimu and Wang, Fengjie and Zhao, Jian},
booktitle={Findings of the Association for Computational Linguistics: ACL 2025},
pages={26584--26606},
year={2025}
}
ELIOT: Zero-Shot Video-Text Retrieval through Relevance-Boosted Captioning and Structural Information Extraction
Xuye Liu, Yimu Wang, Jian Zhao NAACL Student Research Workshop (SRW of NAACL), 2025.
@inproceedings{liu-etal-2025-eliot,
title = "{ELIOT}: Zero-Shot Video-Text Retrieval through Relevance-Boosted Captioning and Structural Information Extraction",
author = "Liu, Xuye and Wang, Yimu and Zhao, Jian",
booktitle = "Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 4: Student Research Workshop)",
year = "2025",
pages = "381--391",
}
DREAM: Improving Video-Text Retrieval Through Relevance-Based Augmentation Using Large Foundation Models
Yimu Wang, Shuai Yuan, Bo Xue, Xiangru Jian, Wei Pang, Mushi Wang, Ning Yu Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics (NAACL), 2025.
@inproceedings{wang-etal-2025-dream,
title = "{DREAM}: Improving Video-Text Retrieval Through Relevance-Based Augmentation Using Large Foundation Models",
author = "Wang, Yimu and Yuan, Shuai and Xue, Bo and Jian, Xiangru and Pang, Wei and Wang, Mushi and Yu, Ning",
booktitle = "Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)",
year = "2025",
pages = "3037--3056",
}
AIDE: Improving 3D Open-Vocabulary Semantic Segmentation by Aligned Vision-Language Learning
Yimu Wang, Krzysztof Czarnecki IEEE Winter Conference on Applications of Computer Vision (WACV), 2025.
@inproceedings{wang2025aide,
title={AiDe: Improving 3D Open-Vocabulary Semantic Segmentation by Aligned Vision-Language Learning},
author={Wang, Yimu and Czarnecki, Krzysztof},
booktitle={2025 IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)},
pages={2674--2685},
year={2025},
organization={IEEE}
}
Conditional Generative Adversarial Network-Assisted System for Radiation-Free Evaluation of Scoliosis Using a Single Smartphone Photograph: A Model Development and Validation Study
Zhong He, Neng Lu, Yi Chen, Elvis Chun-Sing Chui, Zhen Liu, Xiaodong Qin, Jie Li, Shengru Wang, Junlin Yang, Zhiwei Wang, et al. eClinicalMedicine, 2024.
eClinicalMedicine 2024
@article{he2024conditional,
title={Conditional generative adversarial network-assisted system for radiation-free evaluation of scoliosis using a single smartphone photograph: a model development and validation study},
author={He, Zhong and Lu, Neng and Chen, Yi and Chui, Elvis Chun-Sing and Liu, Zhen and Qin, Xiaodong and Li, Jie and Wang, Shengru and Yang, Junlin and Wang, Zhiwei and others},
journal={Eclinicalmedicine},
volume={75},
year={2024},
publisher={Elsevier}
}
NICE: CVPR 2023 Challenge on Zero-Shot Image Captioning
Taehoon Kim, Pyunghwan Ahn, Sangyun Kim, Sihaeng Lee, Mark Marsden, Alessandra Sala, Seung Hwan Kim, Bohyung Han, Kyoung Mu Lee, Honglak Lee, et al. IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPR Workshops), 2024.
CVPR Workshops 2024
@inproceedings{kim2024nice,
title={NICE: CVPR 2023 challenge on zero-shot image captioning},
author={Kim, Taehoon and Ahn, Pyunghwan and Kim, Sangyun and Lee, Sihaeng and Marsden, Mark and Sala, Alessandra and Kim, Seung Hwan and Han, Bohyung and Lee, Kyoung Mu and Lee, Honglak and others},
booktitle={2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)},
pages={7356--7365},
year={2024},
organization={IEEE}
}
Self-Supervised Pretext Tasks for Event Sequence Data from Detecting Misalignment
Yimu Wang, He Zhao, Ruizhi Deng, Frederick Tung, Greg Mori Conference on Neural Information Processing Systems Workshop (NeurIPS workshop), 2024.
@inproceedings{wang2024self,
title={Self-Supervised Pretext Tasks for Event Sequence Data from Detecting Misalignment},
author={Wang, Yimu and Zhao, He and Deng, Ruizhi and Tung, Frederick and Mori, Greg},
booktitle={NeurIPS 2024 Workshop: Self-Supervised Learning-Theory and Practice},
year={2024}
}
Lost Domain Generalization Is a Natural Consequence of Lack of Training Domains
Yimu Wang, Yihan Wu, Hongyang Zhang Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), 2024.
@inproceedings{wang2024lost,
title={Lost domain generalization is a natural consequence of lack of training domains},
author={Wang, Yimu and Wu, Yihan and Zhang, Hongyang},
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
volume={38},
number={14},
pages={15689--15697},
year={2024}
}
Multiobjective Lipschitz Bandits under Lexicographic Ordering
Bo Xue, Ji Cheng, Fei Liu,Yimu Wang, Qingfu Zhang Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), 2024.
@inproceedings{xue2024multiobjective,
title={Multiobjective lipschitz bandits under lexicographic ordering},
author={Xue, Bo and Cheng, Ji and Liu, Fei and Wang, Yimu and Zhang, Qingfu},
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
volume={38},
number={15},
pages={16238--16246},
year={2024}
}
Efficient Algorithms for Generalized Linear Bandits with Heavy-tailed Rewards
Bo Xue, Yimu Wang, Yuanyu Wan, Jinfeng Yi, and Lijun Zhang Conference on Neural Information Processing Systems (NeurIPS), 2023.
@inproceedings{
xue2023efficient,
title={Efficient Algorithms for Generalized Linear Bandits with Heavy-tailed Rewards},
author={Bo Xue and Yimu Wang and Yuanyu Wan and Jinfeng Yi and Lijun Zhang},
booktitle={Thirty-seventh Conference on Neural Information Processing Systems},
year={2023},
url={https://openreview.net/forum?id=Vbm5UCaYeh}
}
Balance Act: Mitigating Hubness in Cross-Modal Retrieval with Query and Gallery Banks
Yimu Wang, Xiangru Jian, Bo Xue Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP Oral), 2023.
@inproceedings{wang-etal-2023-balance,
title = "Balance Act: Mitigating Hubness in Cross-Modal Retrieval with Query and Gallery Banks",
author = "Wang, Yimu and
Jian, Xiangru and
Xue, Bo",
editor = "Bouamor, Houda and
Pino, Juan and
Bali, Kalika",
booktitle = "Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing",
month = dec,
year = "2023",
address = "Singapore",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2023.emnlp-main.652/",
doi = "10.18653/v1/2023.emnlp-main.652",
pages = "10542--10567"
}
Video-Text Retrieval by Supervised Sparse Multi-Grained Learning
Yimu Wang, Peng Shi Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (Findings of EMNLP), 2023.
@inproceedings{wang-shi-2023-video,
title = "Video-Text Retrieval by Supervised Sparse Multi-Grained Learning",
author = "Wang, Yimu and
Shi, Peng",
editor = "Bouamor, Houda and
Pino, Juan and
Bali, Kalika",
booktitle = "Findings of the Association for Computational Linguistics: EMNLP 2023",
month = dec,
year = "2023",
address = "Singapore",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2023.findings-emnlp.46/",
doi = "10.18653/v1/2023.findings-emnlp.46",
pages = "633--649"
}
InvGC: Robust Cross-Modal Retrieval by Inverse Graph Convolution
Xiangru Jian, Yimu Wang Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (Findings of EMNLP), 2023.
@inproceedings{jian-wang-2023-invgc,
title = "{I}nv{GC}: Robust Cross-Modal Retrieval by Inverse Graph Convolution",
author = "Jian, Xiangru and
Wang, Yimu",
editor = "Bouamor, Houda and
Pino, Juan and
Bali, Kalika",
booktitle = "Findings of the Association for Computational Linguistics: EMNLP 2023",
month = dec,
year = "2023",
address = "Singapore",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2023.findings-emnlp.60/",
doi = "10.18653/v1/2023.findings-emnlp.60",
pages = "836--865"
}
Cooperation or Competition: Avoiding Player Domination for Multi-target Robustness by Adaptive Budgets
Yimu Wang, Dinghuai Zhang, Yihan Wu, Heng Huang, Hongyang Zhang IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2023.
@INPROCEEDINGS{10203542,
author={Wang, Yimu and Zhang, Dinghuai and Wu, Yihan and Huang, Heng and Zhang, Hongyang},
booktitle={2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
title={Cooperation or Competition: Avoiding Player Domination for Multi-Target Robustness via Adaptive Budgets},
year={2023},
volume={},
number={},
pages={20564-20574},
keywords={Training;Deep learning;Adaptation models;Computer vision;Games;Benchmark testing;Robustness;Adversarial attack and defense},
doi={10.1109/CVPR52729.2023.01970}}
Investigating the Existence of “Secret Language” in Language Models
@misc{wang2023investigating,
title={Investigating the Existence of Secret Language in Language Models},
author={Yimu Wang and Peng Shi and Hongyang Zhang},
year={2023},
eprint={2307.12507},
archivePrefix={arXiv}
}
Multimodal Federated Learning via Contrastive Representation Ensemble
Qiying Yu, Yang Liu, Yimu Wang, Ke Xu, Jingjing Liu International Conference on Learning Representations (ICLR), 2023.
@inproceedings{
yu2023multimodal,
title={Multimodal Federated Learning via Contrastive Representation Ensemble},
author={Qiying Yu and Yang Liu and Yimu Wang and Ke Xu and Jingjing Liu},
booktitle={The Eleventh International Conference on Learning Representations },
year={2023},
url={https://openreview.net/forum?id=Hnk1WRMAYqg}
}
Multi-View Fusion Transformer for Sensor-Based Human Activity Recognition
@misc{wang2022multiview,
title={Multi-View Fusion Transformer for Sensor-Based Human Activity Recognition},
author={Yimu Wang and Kun Yu and Yubo Wang and Hongwei Xue},
year={2022},
eprint={2202.12949},
archivePrefix={arXiv}
}
Deep Unified Cross-Modality Hashing by Pairwise Data Alignment
Yimu Wang, Bo Xue, Quan Cheng, Yuhui Chen, and Lijun Zhang International Joint Conference on Artificial Intelligence (IJCAI), 2021.
@inproceedings{ijcai2021p156,
title = {Deep Unified Cross-Modality Hashing by Pairwise Data Alignment},
author = {Wang, Yimu and Xue, Bo and Cheng, Quan and Chen, Yuhui and Zhang, Lijun},
booktitle = {Proceedings of the Thirtieth International Joint Conference on
Artificial Intelligence, {IJCAI-21}},
publisher = {International Joint Conferences on Artificial Intelligence Organization},
editor = {Zhi-Hua Zhou},
pages = {1129--1135},
year = {2021},
month = {8},
note = {Main Track},
doi = {10.24963/ijcai.2021/156},
url = {https://doi.org/10.24963/ijcai.2021/156},
}
Classification of Neurofibromatosis-Related Dystrophic or Nondystrophic Scoliosis Based on Image Features Using Bilateral CNN
Zhengwang He, Yimu Wang, Xue Qin, Rui Yin, Yong Qiu, Ke He, Zezhang Zhu Medical Physics, 2021.
Medical Physics 2021
@article{he2021classification,
title={Classification of Neurofibromatosis-Related Dystrophic or Nondystrophic Scoliosis Based on Image Features Using Bilateral CNN},
author={Zhengwang He and Yimu Wang and Xue Qin and Rui Yin and Yong Qiu and Ke He and Zezhang Zhu},
journal={Medical Physics},
volume={48},
number={4},
pages={1571--1583},
year={2021}
}
Piecewise Hashing: A Deep Hashing Method for Large-Scale Fine-Grained Search
Yimu Wang, Xiu-Shen Wei, Bo Xue, Lijun Zhang Chinese Conference on Pattern Recognition and Computer Vision (PRCV), 2020.
PRCV 2020
@inproceedings{wang2020piecewise,
title={Piecewise Hashing: A Deep Hashing Method for Large-Scale Fine-Grained Search},
author={Yimu Wang and Xiu-Shen Wei and Bo Xue and Lijun Zhang},
booktitle={Chinese Conference on Pattern Recognition and Computer Vision},
pages={432--444},
year={2020}
}
Searching Privately by Imperceptible Lying: A Novel Private Hashing Method with Differential Privacy
Yimu Wang, Shiyin Lu, and Lijun Zhang ACM International Conference on Multimedia (ACM MM), 2020.
@inproceedings{10.1145/3394171.3413882,
author = {Wang, Yimu and Lu, Shiyin and Zhang, Lijun},
title = {Searching Privately by Imperceptible Lying: A Novel Private Hashing Method with Differential Privacy},
year = {2020},
isbn = {9781450379885},
publisher = {Association for Computing Machinery},
address = {New York, NY, USA},
url = {https://doi.org/10.1145/3394171.3413882},
doi = {10.1145/3394171.3413882},
abstract = {In the big data era, with the increasing amount of multi-media data, approximate nearest neighbor~(ANN) search has been an important but challenging problem. As a widely applied large-scale ANN search method, hashing has made great progress, and achieved sub-linear search time with low memory space. However, the advances in hashing are based on the availability of large and representative datasets, which often contain sensitive information. Typically, the privacy of this individually sensitive information is compromised. In this paper, we tackle this valuable yet challenging problem and formulate a task termed as private hashing, which takes into account both searching performance and privacy protection. Specifically, we propose a novel noise mechanism, i.e., Random Flipping, and two private hashing algorithms, i.e., PHashing and PITQ, with the refined analysis within the framework of differential privacy, since differential privacy is a well-established technique to measure the privacy leakage of an algorithm. Random Flipping targets binary scenarios and leverages the "Imperceptible Lying" idea to guarantee ε-differential privacy by flipping each datum of the binary matrix (noise addition). To preserve ε-differential privacy, PHashing perturbs and adds noise to the hash codes learned by non-private hashing algorithms using Random Flipping. However, the noise addition for privacy in PHashing will cause severe performance drops. To alleviate this problem, PITQ leverages the power of alternative learning to distribute the noise generated by Random Flipping into each iteration while preserving ε-differential privacy. Furthermore, to empirically evaluate our algorithms, we conduct comprehensive experiments on the image search task and demonstrate that proposed algorithms achieve equal performance compared with non-private hashing methods.},
booktitle = {Proceedings of the 28th ACM International Conference on Multimedia},
pages = {2700–2709},
numpages = {10},
keywords = {large-scale multimedia retrieval, hashing, differential privacy},
location = {Seattle, WA, USA},
series = {MM '20}
}
Nearly Optimal Regret for Stochastic Linear Bandits with Heavy-Tailed Payoffs
Bo Xue, Guanghui Wang, Yimu Wang, Lijun Zhang International Joint Conference on Artificial Intelligence (IJCAI), 2020.
@inproceedings{ijcai2020p406,
title = {Nearly Optimal Regret for Stochastic Linear Bandits with Heavy-Tailed Payoffs},
author = {Xue, Bo and Wang, Guanghui and Wang, Yimu and Zhang, Lijun},
booktitle = {Proceedings of the Twenty-Ninth International Joint Conference on
Artificial Intelligence, {IJCAI-20}},
publisher = {International Joint Conferences on Artificial Intelligence Organization},
editor = {Christian Bessiere},
pages = {2936--2942},
year = {2020},
month = {7},
note = {Main track},
doi = {10.24963/ijcai.2020/406},
url = {https://doi.org/10.24963/ijcai.2020/406},
}
An Adversarial Domain Adaptation Network for Cross-Domain Fine-Grained Recognition
Yimu Wang, Ren-Jie Song, Xiu-Shen Wei, and Lijun Zhang IEEE Winter Conference on Applications of Computer Vision (WACV), 2020.