@article{wu2026adapt,
title={ADAPT: Agile Diffusion Action Priors for Robust and Steerable Online Text-Driven Humanoid Control},
author={Wu, Yan and Li, Chenhao and Zhao, Kaifeng and Li, Gen and Hutter, Marco and Tang, Siyu},
journal={arXiv preprint arXiv:2609.00677},
year={2026}
}
An end-to-end framework for interactive, text-conditioned humanoid control that executes diverse skills, transitions smoothly between prompts, and adapts to downstream tasks.
@inproceedings{chen2026napcontrol,
title={NaP-Control: Navigating Diffusion Prior for Versatile and Fast Character Control},
author={Chen, Chia-Wen and Wu, Yan and Karunratanakul, Korrawe and Tang, Siyu},
booktitle={Proceedings of the European Conference on Computer Vision (ECCV)},
year={2026}
}
A reinforcement-learning approach that steers the latent noise of a diffusion policy prior for fast, robust, and versatile physics-based character control.
@inproceedings{wu2025uniphys,
title={UniPhys: Unified Planner and Controller with Diffusion for Flexible Physics-Based Character Control},
author={Wu, Yan and Karunratanakul, Korrawe and Luo, Zhengyi and Tang, Siyu},
booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
year={2025}
}
A diffusion-based unified planner and controller that generalizes across reactive and long-horizon character-control tasks without task-specific training.
@inproceedings{fu2023sensorimotor,
title={Learning Deep Sensorimotor Policies for Vision-Based Autonomous Drone Racing},
author={Fu, Jiawei and Song, Yunlong and Wu, Yan and Yu, Fisher and Scaramuzza, Davide},
booktitle={IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)},
year={2023}
}
A vision-based sensorimotor policy for autonomous drone racing, learned directly from visual observations and demonstrated in high-speed flight.
@inproceedings{wu2022saga,
title={SAGA: Stochastic Whole-Body Grasping with Contact},
author={Wu, Yan and Wang, Jiahao and Zhang, Yan and Zhang, Siwei and Hilliges, Otmar and Yu, Fisher and Tang, Siyu},
booktitle={European Conference on Computer Vision (ECCV)},
year={2022}
}
Starting from an arbitrary pose, SAGA generates diverse, natural whole-body motions that approach and grasp a target object in 3D space.
@inproceedings{wu2021neural,
title={Neural Architecture Search as Sparse Supernet},
author={Wu, Yan and Liu, Aoming and Huang, Zhiwu and Zhang, Siwei and Van Gool, Luc},
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
year={2021}
}
A continuous architecture representation that frames neural architecture search as learning a sparse supernet.