120 lines
5.4 KiB
Plaintext
120 lines
5.4 KiB
Plaintext
\begin{thebibliography}{17}
|
|
\providecommand{\natexlab}[1]{#1}
|
|
\providecommand{\url}[1]{\texttt{#1}}
|
|
\expandafter\ifx\csname urlstyle\endcsname\relax
|
|
\providecommand{\doi}[1]{doi: #1}\else
|
|
\providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi
|
|
|
|
\bibitem[Black et~al.(2024)Black, Brown, Driess, Esmail, Equi, Finn, Fusai,
|
|
Groom, Hausman, Ichter, et~al.]{black2024pi0}
|
|
K.~Black, N.~Brown, D.~Driess, A.~Esmail, M.~Equi, C.~Finn, N.~Fusai, L.~Groom,
|
|
K.~Hausman, B.~Ichter, et~al.
|
|
\newblock {$\pi_0$}: A vision-language-action flow model for general robot
|
|
control.
|
|
\newblock \emph{arXiv preprint arXiv:2410.24164}, 2024.
|
|
|
|
\bibitem[Chi et~al.(2023)Chi, Xu, Feng, Cousineau, Du, Burchfiel, Tedrake, and
|
|
Song]{chi2023diffusion}
|
|
C.~Chi, Z.~Xu, S.~Feng, E.~Cousineau, Y.~Du, B.~Burchfiel, R.~Tedrake, and
|
|
S.~Song.
|
|
\newblock Diffusion policy: Visuomotor policy learning via action diffusion.
|
|
\newblock \emph{arXiv preprint arXiv:2303.04137}, 2023.
|
|
|
|
\bibitem[Ho et~al.(2020)Ho, Jain, and Abbeel]{ho2020denoising}
|
|
J.~Ho, A.~Jain, and P.~Abbeel.
|
|
\newblock Denoising diffusion probabilistic models.
|
|
\newblock In \emph{Advances in Neural Information Processing Systems}, 2020.
|
|
|
|
\bibitem[Peebles and Xie(2023)]{peebles2023scalable}
|
|
W.~Peebles and S.~Xie.
|
|
\newblock Scalable diffusion models with transformers.
|
|
\newblock In \emph{IEEE/CVF International Conference on Computer Vision}, 2023.
|
|
|
|
\bibitem[Lipman et~al.(2023)Lipman, Chen, Ben-Hamu, Nickel, and
|
|
Le]{lipman2023flow}
|
|
Y.~Lipman, R.~T.~Q. Chen, H.~Ben-Hamu, M.~Nickel, and M.~Le.
|
|
\newblock Flow matching for generative modeling.
|
|
\newblock In \emph{International Conference on Learning Representations}, 2023.
|
|
|
|
\bibitem[Liu et~al.(2022)Liu, Gong, and Liu]{liu2022flow}
|
|
X.~Liu, C.~Gong, and Q.~Liu.
|
|
\newblock Flow straight and fast: Learning to generate and transfer data with
|
|
rectified flow.
|
|
\newblock \emph{arXiv preprint arXiv:2209.03003}, 2022.
|
|
|
|
\bibitem[Geng et~al.(2025{\natexlab{a}})Geng, Deng, Bai, Kolter, and
|
|
He]{geng2025mean}
|
|
Z.~Geng, M.~Deng, X.~Bai, J.~Z. Kolter, and K.~He.
|
|
\newblock Mean flows for one-step generative modeling.
|
|
\newblock \emph{arXiv preprint arXiv:2505.13447}, 2025{\natexlab{a}}.
|
|
|
|
\bibitem[Geng et~al.(2025{\natexlab{b}})Geng, Lu, Wu, Shechtman, Kolter, and
|
|
He]{geng2025improved}
|
|
Z.~Geng, Y.~Lu, Z.~Wu, E.~Shechtman, J.~Z. Kolter, and K.~He.
|
|
\newblock Improved mean flows: On the challenges of fastforward generative
|
|
models.
|
|
\newblock \emph{arXiv preprint arXiv:2512.02012}, 2025{\natexlab{b}}.
|
|
|
|
\bibitem[He et~al.(2016)He, Zhang, Ren, and Sun]{he2016deep}
|
|
K.~He, X.~Zhang, S.~Ren, and J.~Sun.
|
|
\newblock Deep residual learning for image recognition.
|
|
\newblock In \emph{IEEE Conference on Computer Vision and Pattern Recognition},
|
|
2016.
|
|
|
|
\bibitem[Vaswani et~al.(2017)Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez,
|
|
Kaiser, and Polosukhin]{vaswani2017attention}
|
|
A.~Vaswani, N.~Shazeer, N.~Parmar, J.~Uszkoreit, L.~Jones, A.~N. Gomez,
|
|
L.~Kaiser, and I.~Polosukhin.
|
|
\newblock Attention is all you need.
|
|
\newblock In \emph{Advances in Neural Information Processing Systems}, 2017.
|
|
|
|
\bibitem[{Kimi Team}(2026)]{kimi2026attention}
|
|
{Kimi Team}.
|
|
\newblock Attention residuals.
|
|
\newblock \emph{arXiv preprint arXiv:2603.15031}, 2026.
|
|
|
|
\bibitem[Zhao et~al.(2023)Zhao, Kumar, Levine, and Finn]{zhao2023learning}
|
|
T.~Z. Zhao, V.~Kumar, S.~Levine, and C.~Finn.
|
|
\newblock Learning fine-grained bimanual manipulation with low-cost hardware.
|
|
\newblock \emph{arXiv preprint arXiv:2304.13705}, 2023.
|
|
|
|
\bibitem[Shukor et~al.(2025)Shukor, Aubakirova, Capuano, Kooijmans, Palma,
|
|
Zouitine, Aractingi, Pascal, Russi, Marafioti, et~al.]{shukor2025smolvla}
|
|
M.~Shukor, D.~Aubakirova, F.~Capuano, P.~Kooijmans, S.~Palma, A.~Zouitine,
|
|
M.~Aractingi, C.~Pascal, M.~Russi, A.~Marafioti, et~al.
|
|
\newblock Smolvla: A vision-language-action model for affordable and efficient
|
|
robotics.
|
|
\newblock \emph{arXiv preprint arXiv:2506.01844}, 2025.
|
|
|
|
\bibitem[Brohan et~al.(2022)Brohan, Brown, Carbajal, Chebotar, Dabis, Finn,
|
|
Gopalakrishnan, Hausman, Herzog, Hsu, et~al.]{brohan2022rt}
|
|
A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, J.~Dabis, C.~Finn,
|
|
K.~Gopalakrishnan, K.~Hausman, A.~Herzog, J.~Hsu, et~al.
|
|
\newblock Rt-1: Robotics transformer for real-world control at scale.
|
|
\newblock \emph{arXiv preprint arXiv:2212.06817}, 2022.
|
|
|
|
\bibitem[Brohan et~al.(2023)Brohan, Brown, Carbajal, Chebotar, Chen,
|
|
Choromanski, Ding, Driess, Dubey, Finn, et~al.]{brohan2023rt}
|
|
A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, X.~Chen, K.~Choromanski,
|
|
T.~Ding, D.~Driess, A.~Dubey, C.~Finn, et~al.
|
|
\newblock Rt-2: Vision-language-action models transfer web knowledge to robotic
|
|
control.
|
|
\newblock In \emph{Conference on Robot Learning}, 2023.
|
|
|
|
\bibitem[Kim et~al.(2024)Kim, Pertsch, Karamcheti, Xiao, Balakrishna, Nair,
|
|
Rafailov, Foster, Lam, Sanketi, et~al.]{kim2024openvla}
|
|
M.~J. Kim, K.~Pertsch, S.~Karamcheti, T.~Xiao, A.~Balakrishna, S.~Nair,
|
|
R.~Rafailov, E.~Foster, G.~Lam, P.~Sanketi, et~al.
|
|
\newblock Openvla: An open-source vision-language-action model.
|
|
\newblock \emph{arXiv preprint arXiv:2406.09246}, 2024.
|
|
|
|
\bibitem[Pertsch et~al.(2025)Pertsch, Stachowicz, Ichter, Driess, Nair, Vuong,
|
|
Mees, Finn, and Levine]{pertsch2025fast}
|
|
K.~Pertsch, K.~Stachowicz, B.~Ichter, D.~Driess, S.~Nair, Q.~Vuong, O.~Mees,
|
|
C.~Finn, and S.~Levine.
|
|
\newblock Fast: Efficient action tokenization for vision-language-action
|
|
models.
|
|
\newblock \emph{arXiv preprint arXiv:2501.09747}, 2025.
|
|
|
|
\end{thebibliography}
|