141 lines
5.6 KiB
BibTeX
141 lines
5.6 KiB
BibTeX
@article{chi2023diffusion,
|
|
title = {Diffusion Policy: Visuomotor Policy Learning via Action Diffusion},
|
|
author = {Chi, Cheng and Xu, Zhenjia and Feng, Siyuan and Cousineau, Eric and Du, Yilun and Burchfiel, Benjamin and Tedrake, Russ and Song, Shuran},
|
|
year = {2023},
|
|
journal = {arXiv preprint arXiv:2303.04137},
|
|
eprint = {2303.04137},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@inproceedings{ho2020denoising,
|
|
title = {Denoising Diffusion Probabilistic Models},
|
|
author = {Ho, Jonathan and Jain, Ajay and Abbeel, Pieter},
|
|
year = {2020},
|
|
booktitle = {Advances in Neural Information Processing Systems}
|
|
}
|
|
|
|
@inproceedings{peebles2023scalable,
|
|
title = {Scalable Diffusion Models with Transformers},
|
|
author = {Peebles, William and Xie, Saining},
|
|
year = {2023},
|
|
booktitle = {IEEE/CVF International Conference on Computer Vision}
|
|
}
|
|
|
|
@inproceedings{lipman2023flow,
|
|
title = {Flow Matching for Generative Modeling},
|
|
author = {Lipman, Yaron and Chen, Ricky T. Q. and Ben-Hamu, Heli and Nickel, Maximilian and Le, Matt},
|
|
year = {2023},
|
|
booktitle = {International Conference on Learning Representations}
|
|
}
|
|
|
|
@article{liu2022flow,
|
|
title = {Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow},
|
|
author = {Liu, Xingchao and Gong, Chengyue and Liu, Qiang},
|
|
year = {2022},
|
|
journal = {arXiv preprint arXiv:2209.03003},
|
|
eprint = {2209.03003},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{geng2025mean,
|
|
title = {Mean Flows for One-step Generative Modeling},
|
|
author = {Geng, Zhengyang and Deng, Mingyang and Bai, Xingjian and Kolter, J. Zico and He, Kaiming},
|
|
year = {2025},
|
|
journal = {arXiv preprint arXiv:2505.13447},
|
|
eprint = {2505.13447},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{geng2025improved,
|
|
title = {Improved Mean Flows: On the Challenges of Fastforward Generative Models},
|
|
author = {Geng, Zhengyang and Lu, Yiyang and Wu, Zongze and Shechtman, Eli and Kolter, J. Zico and He, Kaiming},
|
|
year = {2025},
|
|
journal = {arXiv preprint arXiv:2512.02012},
|
|
eprint = {2512.02012},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{brohan2022rt,
|
|
title = {RT-1: Robotics Transformer for Real-World Control at Scale},
|
|
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Dabis, Joseph and Finn, Chelsea and Gopalakrishnan, Keerthana and Hausman, Karol and Herzog, Alexander and Hsu, Jasmine and others},
|
|
year = {2022},
|
|
journal = {arXiv preprint arXiv:2212.06817},
|
|
eprint = {2212.06817},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@inproceedings{brohan2023rt,
|
|
title = {RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control},
|
|
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Chen, Xi and Choromanski, Krzysztof and Ding, Tianli and Driess, Danny and Dubey, Avinava and Finn, Chelsea and others},
|
|
year = {2023},
|
|
booktitle = {Conference on Robot Learning}
|
|
}
|
|
|
|
@article{kim2024openvla,
|
|
title = {OpenVLA: An Open-Source Vision-Language-Action Model},
|
|
author = {Kim, Moo Jin and Pertsch, Karl and Karamcheti, Siddharth and Xiao, Ted and Balakrishna, Ashwin and Nair, Suraj and Rafailov, Rafael and Foster, Ethan and Lam, Grace and Sanketi, Pannag and others},
|
|
year = {2024},
|
|
journal = {arXiv preprint arXiv:2406.09246},
|
|
eprint = {2406.09246},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{shukor2025smolvla,
|
|
title = {SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},
|
|
author = {Shukor, Mustafa and Aubakirova, Dana and Capuano, Francesco and Kooijmans, Pepijn and Palma, Steven and Zouitine, Adil and Aractingi, Michel and Pascal, Caroline and Russi, Martino and Marafioti, Andres and others},
|
|
year = {2025},
|
|
journal = {arXiv preprint arXiv:2506.01844},
|
|
eprint = {2506.01844},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{black2024pi0,
|
|
title = {{$\pi_0$}: A Vision-Language-Action Flow Model for General Robot Control},
|
|
author = {Black, Kevin and Brown, Noah and Driess, Danny and Esmail, Adnan and Equi, Michael and Finn, Chelsea and Fusai, Niccolo and Groom, Lachy and Hausman, Karol and Ichter, Brian and others},
|
|
year = {2024},
|
|
journal = {arXiv preprint arXiv:2410.24164},
|
|
eprint = {2410.24164},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{pertsch2025fast,
|
|
title = {FAST: Efficient Action Tokenization for Vision-Language-Action Models},
|
|
author = {Pertsch, Karl and Stachowicz, Kyle and Ichter, Brian and Driess, Danny and Nair, Suraj and Vuong, Quan and Mees, Oier and Finn, Chelsea and Levine, Sergey},
|
|
year = {2025},
|
|
journal = {arXiv preprint arXiv:2501.09747},
|
|
eprint = {2501.09747},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@article{zhao2023learning,
|
|
title = {Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware},
|
|
author = {Zhao, Tony Z. and Kumar, Vikash and Levine, Sergey and Finn, Chelsea},
|
|
year = {2023},
|
|
journal = {arXiv preprint arXiv:2304.13705},
|
|
eprint = {2304.13705},
|
|
archivePrefix = {arXiv}
|
|
}
|
|
|
|
@inproceedings{he2016deep,
|
|
title = {Deep Residual Learning for Image Recognition},
|
|
author = {He, Kaiming and Zhang, Xiangyu and Ren, Shaoqing and Sun, Jian},
|
|
year = {2016},
|
|
booktitle = {IEEE Conference on Computer Vision and Pattern Recognition}
|
|
}
|
|
|
|
@inproceedings{vaswani2017attention,
|
|
title = {Attention Is All You Need},
|
|
author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N. and Kaiser, Lukasz and Polosukhin, Illia},
|
|
year = {2017},
|
|
booktitle = {Advances in Neural Information Processing Systems}
|
|
}
|
|
|
|
@article{kimi2026attention,
|
|
title = {Attention Residuals},
|
|
author = {{Kimi Team}},
|
|
year = {2026},
|
|
journal = {arXiv preprint arXiv:2603.15031},
|
|
eprint = {2603.15031},
|
|
archivePrefix = {arXiv}
|
|
}
|