\begin{thebibliography}{16} \providecommand{\natexlab}[1]{#1} \providecommand{\url}[1]{\texttt{#1}} \expandafter\ifx\csname urlstyle\endcsname\relax \providecommand{\doi}[1]{doi: #1}\else \providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi \bibitem[Chi et~al.(2023)Chi, Xu, Feng, Cousineau, Du, Burchfiel, Tedrake, and Song]{chi2023diffusion} C.~Chi, Z.~Xu, S.~Feng, E.~Cousineau, Y.~Du, B.~Burchfiel, R.~Tedrake, and S.~Song. \newblock Diffusion policy: Visuomotor policy learning via action diffusion. \newblock \emph{arXiv preprint arXiv:2303.04137}, 2023. \bibitem[Ho et~al.(2020)Ho, Jain, and Abbeel]{ho2020denoising} J.~Ho, A.~Jain, and P.~Abbeel. \newblock Denoising diffusion probabilistic models. \newblock In \emph{Advances in Neural Information Processing Systems}, 2020. \bibitem[Peebles and Xie(2023)]{peebles2023scalable} W.~Peebles and S.~Xie. \newblock Scalable diffusion models with transformers. \newblock In \emph{IEEE/CVF International Conference on Computer Vision}, 2023. \bibitem[Lipman et~al.(2023)Lipman, Chen, Ben-Hamu, Nickel, and Le]{lipman2023flow} Y.~Lipman, R.~T.~Q. Chen, H.~Ben-Hamu, M.~Nickel, and M.~Le. \newblock Flow matching for generative modeling. \newblock In \emph{International Conference on Learning Representations}, 2023. \bibitem[Liu et~al.(2022)Liu, Gong, and Liu]{liu2022flow} X.~Liu, C.~Gong, and Q.~Liu. \newblock Flow straight and fast: Learning to generate and transfer data with rectified flow. \newblock \emph{arXiv preprint arXiv:2209.03003}, 2022. \bibitem[Geng et~al.(2025{\natexlab{a}})Geng, Deng, Bai, Kolter, and He]{geng2025mean} Z.~Geng, M.~Deng, X.~Bai, J.~Z. Kolter, and K.~He. \newblock Mean flows for one-step generative modeling. \newblock \emph{arXiv preprint arXiv:2505.13447}, 2025{\natexlab{a}}. \bibitem[Geng et~al.(2025{\natexlab{b}})Geng, Lu, Wu, Shechtman, Kolter, and He]{geng2025improved} Z.~Geng, Y.~Lu, Z.~Wu, E.~Shechtman, J.~Z. Kolter, and K.~He. \newblock Improved mean flows: On the challenges of fastforward generative models. \newblock \emph{arXiv preprint arXiv:2512.02012}, 2025{\natexlab{b}}. \bibitem[He et~al.(2016)He, Zhang, Ren, and Sun]{he2016deep} K.~He, X.~Zhang, S.~Ren, and J.~Sun. \newblock Deep residual learning for image recognition. \newblock In \emph{IEEE Conference on Computer Vision and Pattern Recognition}, 2016. \bibitem[Vaswani et~al.(2017)Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez, Kaiser, and Polosukhin]{vaswani2017attention} A.~Vaswani, N.~Shazeer, N.~Parmar, J.~Uszkoreit, L.~Jones, A.~N. Gomez, L.~Kaiser, and I.~Polosukhin. \newblock Attention is all you need. \newblock In \emph{Advances in Neural Information Processing Systems}, 2017. \bibitem[{Kimi Team}(2026)]{kimi2026attention} {Kimi Team}. \newblock Attention residuals. \newblock \emph{arXiv preprint arXiv:2603.15031}, 2026. \bibitem[Zhao et~al.(2023)Zhao, Kumar, Levine, and Finn]{zhao2023learning} T.~Z. Zhao, V.~Kumar, S.~Levine, and C.~Finn. \newblock Learning fine-grained bimanual manipulation with low-cost hardware. \newblock \emph{arXiv preprint arXiv:2304.13705}, 2023. \bibitem[Shukor et~al.(2025)Shukor, Aubakirova, Capuano, Kooijmans, Palma, Zouitine, Aractingi, Pascal, Russi, Marafioti, et~al.]{shukor2025smolvla} M.~Shukor, D.~Aubakirova, F.~Capuano, P.~Kooijmans, S.~Palma, A.~Zouitine, M.~Aractingi, C.~Pascal, M.~Russi, A.~Marafioti, et~al. \newblock Smolvla: A vision-language-action model for affordable and efficient robotics. \newblock \emph{arXiv preprint arXiv:2506.01844}, 2025. \bibitem[Brohan et~al.(2022)Brohan, Brown, Carbajal, Chebotar, Dabis, Finn, Gopalakrishnan, Hausman, Herzog, Hsu, et~al.]{brohan2022rt} A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, J.~Dabis, C.~Finn, K.~Gopalakrishnan, K.~Hausman, A.~Herzog, J.~Hsu, et~al. \newblock Rt-1: Robotics transformer for real-world control at scale. \newblock \emph{arXiv preprint arXiv:2212.06817}, 2022. \bibitem[Brohan et~al.(2023)Brohan, Brown, Carbajal, Chebotar, Chen, Choromanski, Ding, Driess, Dubey, Finn, et~al.]{brohan2023rt} A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, X.~Chen, K.~Choromanski, T.~Ding, D.~Driess, A.~Dubey, C.~Finn, et~al. \newblock Rt-2: Vision-language-action models transfer web knowledge to robotic control. \newblock In \emph{Conference on Robot Learning}, 2023. \bibitem[Kim et~al.(2024)Kim, Pertsch, Karamcheti, Xiao, Balakrishna, Nair, Rafailov, Foster, Lam, Sanketi, et~al.]{kim2024openvla} M.~J. Kim, K.~Pertsch, S.~Karamcheti, T.~Xiao, A.~Balakrishna, S.~Nair, R.~Rafailov, E.~Foster, G.~Lam, P.~Sanketi, et~al. \newblock Openvla: An open-source vision-language-action model. \newblock \emph{arXiv preprint arXiv:2406.09246}, 2024. \bibitem[Pertsch et~al.(2025)Pertsch, Stachowicz, Ichter, Driess, Nair, Vuong, Mees, Finn, and Levine]{pertsch2025fast} K.~Pertsch, K.~Stachowicz, B.~Ichter, D.~Driess, S.~Nair, Q.~Vuong, O.~Mees, C.~Finn, and S.~Levine. \newblock Fast: Efficient action tokenization for vision-language-action models. \newblock \emph{arXiv preprint arXiv:2501.09747}, 2025. \end{thebibliography}