feat(vla): add SmolVLA conditioning and experiment artifacts

This commit is contained in:
Logic
2026-07-31 10:11:04 +08:00
parent acbd7c605a
commit 5ae9f5fa48
175 changed files with 317471 additions and 70 deletions
+194
View File
@@ -0,0 +1,194 @@
{
"papers": [
{
"bibtex_key": "chi2023diffusion",
"title": "Diffusion Policy: Visuomotor Policy Learning via Action Diffusion",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "ho2020denoising",
"title": "Denoising Diffusion Probabilistic Models",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "peebles2023scalable",
"title": "Scalable Diffusion Models with Transformers",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "lipman2023flow",
"title": "Flow Matching for Generative Modeling",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "liu2022flow",
"title": "Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "geng2025mean",
"title": "Mean Flows for One-step Generative Modeling",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "geng2025improved",
"title": "Improved Mean Flows: On the Challenges of Fastforward Generative Models",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "brohan2022rt",
"title": "RT-1: Robotics Transformer for Real-World Control at Scale",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "brohan2023rt",
"title": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "kim2024openvla",
"title": "OpenVLA: An Open-Source Vision-Language-Action Model",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "shukor2025smolvla",
"title": "SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "black2024pi0",
"title": "$\\\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "pertsch2025fast",
"title": "FAST: Efficient Action Tokenization for Vision-Language-Action Models",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "zhao2023learning",
"title": "Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "he2016deep",
"title": "Deep Residual Learning for Image Recognition",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "vaswani2017attention",
"title": "Attention Is All You Need",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
},
{
"bibtex_key": "kimi2026attention",
"title": "Attention Residuals",
"year": null,
"abstract": "Verified from arXiv/direct web sources during drafting.",
"authors": [
{
"name": "see refs.bib"
}
]
}
],
"min_cite_paper_count": 15,
"n_total": 17,
"note": "Manual citation pool constructed after Exa discovery; Semantic Scholar key was not visible in shell and unauthenticated S2 was rate-limited."
}
+721
View File
@@ -0,0 +1,721 @@
{
"candidates": [
{
"title": "Visuomotor policy learning via action diffusion - ACM Digital Library",
"snippet": "Diffusion policy: : Visuomotor policy learning via action diffusion: International Journal of Robotics Research: Vol 4\n[...]\n, No 1\n[...]\nother-periodical;requested\n[...]\n:string:\n[...]\nPublication Websites;subPage:\n[...]\n:Basic Abstract\n[...]\n;page:string:Article/Chapter View;ctype:string:Journal Content;group\n[...]\n:acm-\n[...]\ntype>other-periodical;website:website:\n[...]\n-site;\n[...]\n:issue:\n[...]\n\\:10.5\n[...]\n55/rbrs.20\n[...]\n.44.issue-10-11;csubtype:string:Periodical;taxonomy:taxonomy:acm-pubtype;pageGroup:string:Publication Pages\"> skip to main content\n\n \n\n \n\nContents\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion mode",
"source_url": "https://dl.acm.org/doi/10.1177/02783649241273668",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://dl.acm.org/doi/10.1177/02783649241273668",
"_exa_published_date": null
},
{
"title": "Visuomotor Policy Learning via Action Diffusion - arXiv",
"snippet": "# Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details is available diffusion-policy.cs.columbia.edu\n[...]\nIn this work, we s",
"source_url": "https://arxiv.org/abs/2303.04137",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2303.04137",
"_exa_published_date": "2023-03-07T00:00:00.000Z"
},
{
"title": "[2303.04137v5] Diffusion Policy: Visuomotor Policy Learning via Action Diffusion",
"snippet": "[2303.04137v5] Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\n# Title:Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\n> Abstract:This paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robot's visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models.",
"source_url": "http://arxiv.org/abs/2303.04137v5",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "http://arxiv.org/abs/2303.04137v5",
"_exa_published_date": null
},
{
"title": "[2303.04137] Diffusion Policy - ar5iv - arXiv",
"snippet": "# Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details will be publicly available.\n[...]\nIn this work, we seek to address thi",
"source_url": "https://ar5iv.labs.arxiv.org/html/2303.04137",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://ar5iv.labs.arxiv.org/html/2303.04137",
"_exa_published_date": null
},
{
"title": "[2006.11239v2] Denoising Diffusion Probabilistic Models",
"snippet": "[2006.11239\n[...]\n] Denoising Diffusion Probabilistic Models\n[...]\n# Title:Denoising Diffusion Probabilistic Models\n[...]\nAuthors: Jonathan Ho, Ajay Jain, Pieter Abbeel\n[...]\n> Abstract:We present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our implementation is available at this https URL\n\n \n\nhttps://doi.org/10.48550/arXiv.2006.11239\n\n \n\narXiv-issued DOI via DataCite",
"source_url": "https://arxiv.org/abs/2006.11239v2",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2006.11239v2",
"_exa_published_date": null
},
{
"title": "[2006.11239] Denoising Diffusion Probabilistic Models - arXiv",
"snippet": "[2006.11239] Denoising Diffusion Probabilistic Models\n[...]\n# Denoising Diffusion Probabilistic Models\n[...]\nJonathan Ho UC Berkeley jonathanho@berkeley.edu &Ajay Jain UC Berkeley ajayj@berkeley.edu &Pieter Abbeel UC Berkeley pabbeel@cs.berkeley.edu\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our implementation is available at https://github.com/hojonathanho/diffusion.\n[...]\nThis paper presents progress in diffusion probabilistic models [53]. A diffusion probabilistic model (which we will call a “diffusion model” for brevity) is a parameterized Markov chain trained using variational inference to produce samples matching the data after finite time. Transitions of this chain are learned to reverse a diffusion process, which is a Markov chain that gradually adds noise to the data in the opposite direction of s",
"source_url": "https://arxiv.org/abs/2006.11239",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2006.11239",
"_exa_published_date": "2020-06-19T00:00:00.000Z"
},
{
"title": "Denoising Diffusion Probabilistic Models",
"snippet": "Denoising Diffusion Probabilistic Models \n\nAuthorFeedback Bibtex MetaReview Paper Review Supplemental\n\n## Abstract\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN.\n\n \n\nDo not remove: This comment is monitored to verify that the site is working properly",
"source_url": "https://papers.nips.cc/paper/2020/hash/4c5bcfec8584af0d967f1ab10179ca4b-Abstract.html",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://papers.nips.cc/paper/2020/hash/4c5bcfec8584af0d967f1ab10179ca4b-Abstract.html",
"_exa_published_date": null
},
{
"title": "",
"snippet": "Denoising Diffusion Probabilistic Models\n[...]\njonathanho@berkeley.edu\n[...]\nAjay Jain\n[...]\najayj@berkeley.edu\n[...]\nPieter Abbeel\n[...]\npabbeel@cs.berkeley.edu\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models,\n[...]\na class of latent variable models inspired by considerations from nonequilibrium\n[...]\nthermodynamics. Our best results are obtained by training on a weighted variational\n[...]\nbound designed according to a novel connection between diffusion probabilistic\n[...]\nmodels and denoising score matching with Langevin dynamics, and our models nat\u0002urally admit a progressive lossy decompression scheme that can be interpreted as a\n[...]\ngeneralization of autoregressive decoding. On the unconditional CIFAR10 dataset,\n[...]\nwe obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On\n[...]\n256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our imple\u0002mentation is available at https://github.com/hojonathanho/diffusion.\n[...]\nThis paper presents progress in diffusion probabilistic models [50]. A diffusion probabilistic model\n[...]\n(which we will call a “diffusion model” for brevity) is a parameterized Markov chain trained using\n[...]\nvariational inference to produce samples matching the data after finite time. Transitions of this chain\n[...]\nare learned to reverse a diffusion process, which is a Markov chain that gradually adds noise to the\n[...]\ndata in the opposite direction of sampling until signal",
"source_url": "https://arxiv.org/pdf/2006.11239v1",
"discovered_for": [
"rw.diffusion_policy",
"rw.mean_flow",
"rw.vla",
"rw.attnres",
"method.training"
],
"_exa_id": "https://arxiv.org/pdf/2006.11239v1",
"_exa_published_date": null
},
{
"title": "[PDF] Denoising Diffusion Probabilistic Models - arXiv",
"snippet": "Denoising Diffusion Probabilistic Models\n[...]\njonathanho@berkeley.edu\n[...]\nAjay Jain\n[...]\najayj@berkeley.edu\n[...]\nPieter Abbeel\n[...]\npabbeel@cs.berkeley.edu\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models,\n[...]\na class of latent variable models inspired by considerations from nonequilibrium\n[...]\nthermodynamics. Our best results are obtained by training on a weighted variational\n[...]\nbound designed according to a novel connection between diffusion probabilistic\n[...]\nmodels and denoising score matching with Langevin dynamics, and our models nat\u0002urally admit a progressive lossy decompression scheme that can be interpreted as a\n[...]\ngeneralization of autoregressive decoding. On the unconditional CIFAR10 dataset,\n[...]\nwe obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On\n[...]\n256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our imple\u0002mentation is available at https://github.com/hojonathanho/diffusion.\n[...]\nThis paper presents progress in diffusion probabilistic models [53]. A diffusion probabilistic model\n[...]\n(which we will call a “diffusion model” for brevity) is a parameterized Markov chain trained using\n[...]\nvariational inference to produce samples matching the data after finite time. Transitions of this chain\n[...]\nare learned to reverse a diffusion process, which is a Markov chain that gradually adds noise to the\n[...]\ndata in the opposite direction of sampling until signal",
"source_url": "https://arxiv.org/pdf/2006.11239",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/pdf/2006.11239",
"_exa_published_date": "2020-12-16T00:00:00.000Z"
},
{
"title": "[2212.09748] Scalable Diffusion Models with Transformers - arXiv",
"snippet": "William Peebles* UC Berkeley Saining Xie New York University\n[...]\nWe explore a new class of diffusion models based on the transformer architecture. We train latent diffusion models of images, replacing the commonly-used U-Net backbone with a transformer that operates on latent patches. We analyze the scalability of our Diffusion Transformers (DiTs) through the lens of forward pass complexity as measured by Gflops. We find that DiTs with higher Gflops—through increased transformer depth/width or increased number of input tokens—consistently have lower FID. In addition to possessing good scalability properties, our largest DiT-XL/2 models outperform all prior diffusion models on the class-conditional ImageNet 512 $\\times$ 512 and 256 $\\times$ 256 benchmarks, achieving a state-of-the-art FID of 2.27 on the latter.\n[...]\n, or Di\n[...]\nfor short. Di\n[...]\nadhere to the best practices of Vision Transformers (ViTs) [10], which have been shown to scale more effectively for visual recognition than\n[...]\nconvolutional networks (e.g., Res\n[...]\n[15]).\n[...]\nMore specifically, we study the scaling behavior of transformers with respect to network complexity vs. sample quality. We show that by constructing and benchmarking the DiT design space under the Latent Diffusion Models (LDMs) [48] framework, where diffusion models are trained within a VAEs latent space, we can successfully replace the U-Net backbone with a transformer. We further show that DiTs are scalable architectures for diff",
"source_url": "https://arxiv.org/abs/2212.09748",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2212.09748",
"_exa_published_date": "2022-12-19T00:00:00.000Z"
},
{
"title": "Scalable Diffusion Models with Transformers | IEEE Conference Publication | IEEE Xplore",
"snippet": "Scalable Diffusion Models with Transformers | IEEE Conference Publication | IEEE Xplore\n\n \n\n \n\n \n\n### IEEE Account",
"source_url": "https://ieeexplore.ieee.org/document/10377858/",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://ieeexplore.ieee.org/document/10377858/",
"_exa_published_date": "2025-05-14T00:00:00.000Z"
},
{
"title": "[2212.09748v1] Scalable Diffusion Models with Transformers",
"snippet": "[2212.09748v1] Scalable Diffusion Models with Transformers\n[...]\n# Title:Scalable Diffusion Models with Transformers\n[...]\nAuthors: William Peebles, Saining Xie\n[...]\n> Abstract:We explore a new class of diffusion models based on the transformer architecture. We train latent diffusion models of images, replacing the commonly-used U-Net backbone with a transformer that operates on latent patches. We analyze the scalability of our Diffusion Transformers (DiTs) through the lens of forward pass complexity as measured by Gflops. We find that DiTs with higher Gflops -- through increased transformer depth/width or increased number of input tokens -- consistently have lower FID. In addition to possessing good scalability properties, our largest DiT-XL/2 models outperform all prior diffusion models on the class-conditional ImageNet 512x512 and 256x256 benchmarks, achieving a state-of-the-art FID of 2.27 on the latter.",
"source_url": "http://arxiv.org/abs/2212.09748v1",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "http://arxiv.org/abs/2212.09748v1",
"_exa_published_date": null
},
{
"title": "[PDF] Scalable Diffusion Models with Transformers | Semantic Scholar",
"snippet": "[PDF] Scalable Diffusion Models with Transformers | Semantic Scholar \n\nNavigate Paper Download (opens in a new tab) Share\n[...]\n```\n@article{Peebles2022ScalableDM,\n title={Scalable Diffusion Models with Transformers},\n author={William S. Peebles and Saining Xie},\n journal={2023 IEEE/CVF International Conference on Computer Vision (ICCV)},\n year={2022},\n pages={4172-4182},\n url={https://api.semanticscholar.org/CorpusID:254854389}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/736973165f98105fec3729b7db414ae4d80fcbeb",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://www.semanticscholar.org/reader/736973165f98105fec3729b7db414ae4d80fcbeb",
"_exa_published_date": "2022-12-19T14:39:50.000Z"
},
{
"title": "Scalable Diffusion Models with Transformers - IEEE Xplore",
"snippet": "Scalable Diffusion Models with Transformers | IEEE Conference Publication | IEEE Xplore\n**\n\n### IEEE Account\n* Change Username/Password\n* Update Address\n### Purchase Details\n* Payment Options\n* Order History\n* View Purchased Documents\n### Profile Information\n* Communications Preferences\n* Profession and Education\n* Technical Interests\n### Need Help?\n* **US & Canada:**+1 800 678 4333\n* **Worldwide:**+1 732 981 0060\n* Contact & Support\n* About IEEE*Xplore*\n* Contact Us\n* Help\n* Accessibility\n* Terms of Use\n* Nondiscrimination Policy\n* Sitemap\n* Privacy & Opting Out of Cookies\nA not-for-profit organization, IEEE is the world's largest technical professional organization dedicated to advancing technology for the benefit of humanity.\n© Copyright 2025 IEEE - All rights reserved. Use of this web site signifies your agreement to the terms and conditions.\n**",
"source_url": "https://ieeexplore.ieee.org/iel7/10376473/10376477/10377858.pdf",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://ieeexplore.ieee.org/iel7/10376473/10376477/10377858.pdf",
"_exa_published_date": null
},
{
"title": "[PDF] Flow Matching for Generative Modeling - arXiv",
"snippet": "FLOW MATCHING FOR GENERATIVE MODELING\n[...]\nYaron Lipman1,2 Ricky T. Q. Chen1 Heli Ben-Hamu2 Maximilian Nickel1 Matt Le1\n[...]\nWe introduce a new paradigm for generative modeling built on Continuous\n[...]\n(CNFs), allowing us to train CNFs at unprecedented scale.\n[...]\nSpecifically, we present the notion of Flow Matching (FM), a simulation-free\n[...]\napproach for training CNFs based on regressing vector fields of fixed conditional\n[...]\nprobability paths. Flow Matching is compatible with a general family of Gaussian\n[...]\nprobability paths for transforming between noise and data samples—which\n[...]\nsubsumes existing diffusion paths as specific instances. Interestingly, we find\n[...]\nthat employing FM with diffusion paths results in a more robust and stable\n[...]\nalternative for training diffusion models. Furthermore, Flow Matching opens\n[...]\nthe door to training CNFs with other, non-diffusion probability paths. An\n[...]\ninstance of particular interest is using Optimal Transport (OT) displacement\n[...]\ninterpolation to define the conditional probability paths. These paths are more\n[...]\nefficient than diffusion paths, provide faster training and sampling, and result in\n[...]\nbetter generalization. Training CNFs using Flow Matching on ImageNet leads\n[...]\nto consistently better performance than alternative diffusion-based methods in\n[...]\nterms of both likelihood and sample quality, and allows fast and reliable sample\n[...]\ngeneration using off-the-shelf numerical ODE solvers.\n",
"source_url": "https://arxiv.org/pdf/2210.02747",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/pdf/2210.02747",
"_exa_published_date": "2023-02-08T00:00:00.000Z"
},
{
"title": "[2210.02747] Flow Matching for Generative Modeling - arXiv",
"snippet": "[221\n[...]\n] Flow Matching for Generative Modeling\n[...]\n# Flow Matching for Generative Modeling\n[...]\nYaron Lipman1,2 Ricky T. Q. Chen1 Heli Ben-Hamu2 Maximilian Nickel1 Matt Le1 1Meta AI (FAIR) 2Weizmann Institute of Science\n[...]\nWe introduce a new paradigm for generative modeling built on Continuous Normalizing Flows (CNFs), allowing us to train CNFs at unprecedented scale. Specifically, we present the notion of Flow Matching (FM), a simulation-free approach for training CNFs based on regressing vector fields of fixed conditional probability paths. Flow Matching is compatible with a general family of Gaussian probability paths for transforming between noise and data samples—which subsumes existing diffusion paths as specific instances. Interestingly, we find that employing FM with diffusion paths results in a more robust and stable alternative for training diffusion models. Furthermore, Flow Matching opens the door to training CNFs with other, non-diffusion probability paths. An instance of particular interest is using Optimal Transport (OT) displacement interpolation to define the conditional probability paths. These paths are more efficient than diffusion paths, provide faster training and sampling, and result in better generalization. Training CNFs using Flow Matching on ImageNet leads to consistently better performance than alternative diffusion-based methods in terms of both likelihood and sample quality, and allows fast and reliable sample generation using off-the-s",
"source_url": "https://arxiv.org/abs/2210.02747",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2210.02747",
"_exa_published_date": "2022-10-06T00:00:00.000Z"
},
{
"title": "[PDF] Flow Matching for Generative Modeling | Semantic Scholar",
"snippet": "```\n@article{Lipman2022FlowMF,\n title={Flow Matching for Generative Modeling},\n author={Yaron Lipman and Ricky T. Q. Chen and Heli Ben-Hamu and Maximilian Nickel and Matt Le},\n journal={ArXiv},\n year={2022},\n volume={abs/2210.02747},\n url={https://api.semanticscholar.org/CorpusID:252734897}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/af68f10ab5078bfc519caae377c90ee6d9c504e9",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://www.semanticscholar.org/reader/af68f10ab5078bfc519caae377c90ee6d9c504e9",
"_exa_published_date": "2022-10-06T03:39:18.000Z"
},
{
"title": "Flow Matching for Generative Modeling - OpenReview",
"snippet": "Flow Matching for Generative Modeling | OpenReview\n\n## Flow Matching for Generative Modeling\n\nICLR 2023 notable top 25%Readers: Everyone\n\nKeywords: continuous normalizing flows, generative models\n\nAbstract: We introduce a new paradigm for generative modeling built on Continuous Normalizing Flows (CNFs), allowing us to train CNFs at unprecedented scale. Specifically, we present the notion of Flow Matching (FM), a simulation-free approach for training CNFs based on regressing vector fields of fixed conditional probability paths. Flow Matching is compatible with a general family of Gaussian probability paths for transforming between noise and data samples---which subsumes existing diffusion paths as specific instances. Interestingly, we find that employing FM with diffusion paths results in a more robust and stable alternative for training diffusion models. Furthermore, Flow Matching opens the door to training CNFs with other, non-diffusion probability paths. An instance of particular interest is using Optimal Transport (OT) displacement interpolation to define the conditional probability paths. These paths are more efficient than diffusion paths, provide faster training and sampling, and result in better generalization. Training CNFs using Flow Matching on ImageNet leads to consistently better performance than alternative diffusion-based methods in terms of both likelihood and sample quality, and allows fast and reliable sample generation using off-the-shelf numerical ODE solve",
"source_url": "https://openreview.net/forum?id=PqvMRDCJT9t",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://openreview.net/forum?id=PqvMRDCJT9t",
"_exa_published_date": "2022-09-29T08:12:00.000Z"
},
{
"title": "[2505.13447] Mean Flows for One-step Generative Modeling - arXiv",
"snippet": "# Mean Flows for One-step Generative Modeling\n[...]\nZhengyang Geng1 Mingyang Deng2 Xingjian Bai2 J. Zico Kolter1 Kaiming He2 1CMU 2MIT Work partly done when visiting MIT.\n[...]\nWe propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the MeanFlow model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256 $\\times$ 256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.\n[...]\nIn this work, we propose a principled and effective framework, termed MeanFlow, for one-step generation. The core idea is to introduce a new ground-truth field representing the average velocity, in contrast to the instantaneous velocity typically modeled in Flow Matching. Average velocity is defined as the ratio of displacement to a time interval, with displacement give",
"source_url": "https://arxiv.org/abs/2505.13447",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2505.13447",
"_exa_published_date": "2025-05-19T00:00:00.000Z"
},
{
"title": "Mean Flows for One-step Generative Modeling",
"snippet": "# Mean Flows for One-step Generative Modeling\n[...]\nZhengyang Geng1 Mingyang Deng2 Xingjian Bai2 J. Zico Kolter1 Kaiming He2 1CMU 2MIT Work partly done when visiting MIT.\n[...]\nWe propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the MeanFlow model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256 $\\times$ 256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.\n[...]\nIn this work, we propose a principled and effective framework, termed MeanFlow, for one-step generation. The core idea is to introduce a new ground-truth field representing the average velocity, in contrast to the instantaneous velocity typically modeled in Flow Matching. Average velocity is defined as the ratio of displacement to a time interval, with displacement give",
"source_url": "https://arxiv.org/html/2505.13447",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/html/2505.13447",
"_exa_published_date": null
},
{
"title": "Mean Flows for One-step Generative Modeling - OpenReview",
"snippet": "Mean Flows for One-step Generative Modeling | OpenReview\n\n## Mean Flows for One-step Generative Modeling\n\n### Zhengyang Geng, Mingyang Deng, Xingjian Bai, J Zico Kolter, Kaiming He\n\nNeurIPS 2025 oralEveryone Revisions BibTeX CC BY 4.0\n\nKeywords: Generative Models\n\nAbstract: We propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the \\textit{MeanFlow} model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256$\\times$256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.\n\nPrimary Area: Deep learning (e.g., architectures, generative models, optimization for deep networks, foundation models, LLMs)\n\nSubmission Number: 754\n\nLoading",
"source_url": "https://openreview.net/forum?id=uWj4s7rMnR",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://openreview.net/forum?id=uWj4s7rMnR",
"_exa_published_date": "2025-10-29T14:53:17.000Z"
},
{
"title": "[PDF] Mean Flows for One-step Generative Modeling | Semantic Scholar",
"snippet": "[PDF] Mean Flows for One-step Generative Modeling | Semantic Scholar\n[...]\n```\n@article{Geng2025MeanFF,\n title={Mean Flows for One-step Generative Modeling},\n author={Zhengyang Geng and Mingyang Deng and Xingjian Bai and J. Zico Kolter and Kaiming He},\n journal={ArXiv},\n year={2025},\n volume={abs/2505.13447},\n url={https://api.semanticscholar.org/CorpusID:278769814}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/19df654b0d0f634a451564346a09af8bd348dac0",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://www.semanticscholar.org/reader/19df654b0d0f634a451564346a09af8bd348dac0",
"_exa_published_date": "2025-05-19T06:36:24.000Z"
},
{
"title": "Improved Mean Flows: On the Challenges of Fastforward Generative ...",
"snippet": "# Improved Mean Flows: On the Challenges of Fastforward Generative Models\n[...]\nZhengyang Geng1,2,3, Yiyang Lu4,2, Zongze Wu3 Eli Shechtman3 J. Zico Kolter1 Kaiming He2 1CMU 2MIT 3Adobe 4THU Equal contribution. Part of this work was done when Z. Geng was interning at Adobe and MIT, and when Y. Lu was interning at MIT.\n[...]\nMeanFlow (MF) has recently been established as a framework for one-step generative modeling. However, its “fastforward” nature introduces key challenges in both the training objective and the guidance mechanism. First, the original MFs training target depends not only on the underlying ground-truth fields but also on the network itself. To address this issue, we recast the objective as a loss on the instantaneous velocity $v$ , re-parameterized by a network that predicts the average velocity $u$ . Our reformulation yields a more standard regression problem and improves the training stability. Second, the original MF fixes the classifier-free guidance scale during training, which sacrifices flexibility. We tackle this issue by formulating guidance as explicit conditioning variables, thereby retaining flexibility at test time. The diverse conditions are processed through in-context conditioning, which reduces model size and benefits performance. Overall, our improved MeanFlow (iMF) method, trained entirely from scratch, achieves 1.72 FID with a single function evaluation (1-NFE) on ImageNet 256 $\\times$ 256. iMF substantially outperforms prior methods of t",
"source_url": "https://arxiv.org/abs/2512.02012",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2512.02012",
"_exa_published_date": "2025-12-01T00:00:00.000Z"
},
{
"title": "[2511.19065] Understanding, Accelerating, and Improving MeanFlow Training",
"snippet": "Accelerating, and Improving MeanFlow Training\n[...]\n# Title:Understanding, Accelerating, and Improving MeanFlow Training\n\nAuthors: Jin-Young Kim, Hyojun Go, Lea Bogensperger, Julius Erbach, Nikolai Kalischek, Federico Tombari, Konrad Schindler, Dominik Narnhofer\n\nView PDF HTML (experimental)\n[...]\n> Abstract:MeanFlow promises high-quality generative modeling in few steps, by jointly learning instantaneous and average velocity fields. Yet, the underlying training dynamics remain unclear. We analyze the interaction between the two velocities and find: (i) well-established instantaneous velocity is a prerequisite for learning average velocity; (ii) learning of instantaneous velocity benefits from average velocity when the temporal gap is small, but degrades as the gap increases; and (iii) task-affinity analysis indicates that smooth learning of large-gap average velocities, essential for one-step generation, depends on the prior formation of accurate instantaneous and small-gap average velocities. Guided by these observations, we design an effective training scheme that accelerates the formation of instantaneous velocity, then shifts emphasis from short- to long-interval average velocity. Our enhanced MeanFlow training yields faster convergence and significantly better few-step generation: With the same DiT-XL backbone, our method reaches an impressive FID of 2.87 on 1-NFE ImageNet 256x256, compared to 3.43 for the conventional MeanFlow baseline. Alternatively, our method matche",
"source_url": "https://arxiv.org/abs/2511.19065",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2511.19065",
"_exa_published_date": null
},
{
"title": "RT-1: Robotics Transformer for Real-World Control at Scale",
"snippet": "By transferring knowledge from large, diverse, task-agnostic datasets, modern machine learning models can solve specific downstream tasks either zero-shot or with small task-specific datasets to a high level of performance. While this capability has been demonstrated in other fields such as computer vision, natural language processing or speech recognition, it remains to be shown in robotics, where the generalization capabilities of the models are particularly critical due to the difficulty of collecting real-world robotic data. We argue that one of the keys to the success of such general robotic models lies with open-ended task-agnostic training, combined with high-capacity architectures that can absorb all of the diverse, robotic data. In this paper, we present a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties. We verify our conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks. The projects website and videos can be found at robotics-transformer1.github.io\n[...]\nThe second challenge lies in the design of the model itself. Effective robotic multi-task learning requires a high capacity model, and Transformer (Vaswani et al., 2017) models excel in this regard, particularly when it is necessary to learn many tasks conditioned, as in our case, on language instruct",
"source_url": "https://arxiv.org/html/2212.06817",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2212.06817",
"_exa_published_date": null
},
{
"title": "RT-1: Robotics Transformer for Real-World Control at Scale - arXiv",
"snippet": "By transferring knowledge from large, diverse, task-agnostic datasets, modern machine learning models can solve specific downstream tasks either zero-shot or with small task-specific datasets to a high level of performance. While this capability has been demonstrated in other fields such as computer vision, natural language processing or speech recognition, it remains to be shown in robotics, where the generalization capabilities of the models are particularly critical due to the difficulty of collecting real-world robotic data. We argue that one of the keys to the success of such general robotic models lies with open-ended task-agnostic training, combined with high-capacity architectures that can absorb all of the diverse, robotic data. In this paper, we present a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties. We verify our conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks. The projects website and videos can be found at robotics-transformer1.github.io\n[...]\nThe second challenge lies in the design of the model itself. Effective robotic multi-task learning requires a high capacity model, and Transformer (Vaswani et al., 2017) models excel in this regard, particularly when it is necessary to learn many tasks conditioned, as in our case, on language instruct",
"source_url": "https://arxiv.org/abs/2212.06817",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2212.06817",
"_exa_published_date": "2022-12-13T00:00:00.000Z"
},
{
"title": "Bringing the RT-1-X Foundation Model to a SCARA robot",
"snippet": "Traditional robotic systems require specific training data for each task, environment, and robot form. While recent advancements in machine learning have enabled models to generalize across new tasks and environments, the challenge of adapting these models to entirely new settings remains largely unexplored. This study addresses this by investigating the generalization capabilities of the RT-1-X robotic foundation model to a type of robot unseen during its training: a SCARA robot from UMI-RTX.\n[...]\nRecent breakthroughs in machine learning and artificial intelligence suggest that training on large, diverse datasets can lead to highly\n[...]\nwhich often exceed\n[...]\nperformance of models developed for\n[...]\ndatasets tailored to\n[...]\nAs a result,\n[...]\nfield has been exploring more\n[...]\ncan adapt to\n[...]\n. Recent advancements like transformer\n[...]\ns RT-1 [brohan_rt-1_2022], which demonstrate the potential for\n[...]\nGoogles RT-1 model is an impressive work, tested on a collection of real-world robotic experiences, where in different institutes a fleet of robots were performing 700 tasks [brohan_rt-1_2022]. The robots in the training set, such as the Franka, Kuka iiwa, UR5 and the EveryDay robot, can move their end-effector in a spherical working-space. None of the robots in the dataset is of the SCARA (Selective Compliance Assembly Robot Arm) type. With a SCARA robot the movement of z-axis is decoupled from the movement in the x-y plane, which gives a SCARA robot an kidney ",
"source_url": "https://arxiv.org/html/2409.03299v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2409.03299v1",
"_exa_published_date": null
},
{
"title": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control | OpenReview",
"snippet": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control | OpenReview\n[...]\n## RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control\n[...]\nTL;DR: Vision-language models, trained on Internet-scale data, can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning.\n[...]\nAbstract: We study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows th",
"source_url": "https://openreview.net/forum?id=XMQgwiJ7KSX",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=XMQgwiJ7KSX",
"_exa_published_date": "2023-08-30T15:38:01.000Z"
},
{
"title": "[PDF] RT-2: Vision-Language-Action Models Transfer Web Knowledge to ...",
"snippet": "Transfer Web Knowledge\n[...]\n# RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control\n[...]\nWe study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows that our approach leads to performant robotic policies and enables RT-2 to obtain a range of emergent capabilities from Internet-scale training. This includes significantly improved generalization to novel objects, the ability to interpret commands not present in the robot t",
"source_url": "https://arxiv.org/pdf/2307.15818",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2307.15818",
"_exa_published_date": "2023-07-28T00:00:00.000Z"
},
{
"title": "[2307.15818] RT-2: Vision-Language-Action Models Transfer Web ...",
"snippet": "Transfer Web Knowledge\n[...]\n# RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control\n[...]\nWe study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows that our approach leads to performant robotic policies and enables RT-2 to obtain a range of emergent capabilities from Internet-scale training. This includes significantly improved generalization to novel objects, the ability to interpret commands not present in the robot t",
"source_url": "https://arxiv.org/abs/2307.15818",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2307.15818",
"_exa_published_date": "2023-07-28T00:00:00.000Z"
},
{
"title": "OpenVLA: An Open-Source Vision-Language-Action Model - arXiv",
"snippet": "OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\n# OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\nLarge policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challenges, we introduce OpenVLA, a 7B-parameter open-source VLA trained on a diverse collection of 970k real-world robot demonstrations. OpenVLA builds on a Llama 2 language model combined with a visual encoder that fuses pretrained features from DINOv2 and SigLIP. As a product of the added data diversity and new model components, OpenVLA demonstrates strong results for generalist manipulation, outperforming closed models such as RT-2-X (55B) by 16.5% in absolute task success rate across 29 tasks and multiple robot embodiments, with 7x fewer parameters. We further show that we can effectively fine-tune OpenVLA for new settings, with especially strong generalization results in multi-task environments involving multiple objects and strong language",
"source_url": "https://arxiv.org/abs/2406.09246",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2406.09246",
"_exa_published_date": "2024-06-13T00:00:00.000Z"
},
{
"title": "[PDF] OpenVLA: An Open-Source Vision-Language-Action Model | Semantic Scholar",
"snippet": "[PDF] OpenVLA: An Open-Source Vision-Language-Action Model | Semantic Scholar \n\nNavigate Paper Download (opens in a new tab) Share\n[...]\n```\n@article{Kim2024OpenVLAAO,\n title={OpenVLA: An Open-Source Vision-Language-Action Model},\n author={Moo Jin Kim and Karl Pertsch and Siddharth Karamcheti and Ted Xiao and Ashwin Balakrishna and Suraj Nair and Rafael Rafailov and Ethan Paul Foster and Grace Lam and Pannag R. Sanketi and Quan Vuong and Thomas Kollar and Benjamin Burchfiel and Russ Tedrake and Dorsa Sadigh and Sergey Levine and Percy Liang and Chelsea Finn},\n journal={ArXiv},\n year={2024},\n volume={abs/2406.09246},\n url={https://api.semanticscholar.org/CorpusID:270440391}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/8f9ceb5ffad8e7a066dfc9d9aaa5153b714740ee",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://www.semanticscholar.org/reader/8f9ceb5ffad8e7a066dfc9d9aaa5153b714740ee",
"_exa_published_date": "2024-06-13T09:02:12.000Z"
},
{
"title": "OpenVLA: An Open-Source Vision-Language-Action Model",
"snippet": "OpenVLA: An Open-Source Vision-Language-Action Model | OpenReview\n\n## OpenVLA: An Open-Source Vision-Language-Action Model\n\n### Moo Jin Kim, Karl Pertsch, Siddharth Karamcheti, Ted Xiao, Ashwin Balakrishna, Suraj Nair, Rafael Rafailov, Ethan P Foster, Pannag R Sanketi, Quan Vuong, Thomas Kollar, Benjamin Burchfiel, Russ Tedrake, Dorsa Sadigh, Sergey Levine, Percy Liang, Chelsea Finn \n\nCoRL 2024everyonesince 05 Sept 2024\">Everyone Revisions BibTeX CC BY 4.0\n\nKeywords: Vision-Language-Action Models, Generalist Policies, Large-scale Robot Learning, Robotic Manipulation, Robotics, Vision-Language Models\n\nTL;DR: We introduce OpenVLA, a state-of-the-art, open-source 7B-parameter VLA model that obtains strong performance for cross-embodiment robot control out-of-the-box and can be easily adapted to new robot setups via parameter-efficient fine-tuning.\n\nAbstract: Large policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challeng",
"source_url": "https://openreview.net/forum?id=ZMnD6QZAE6",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=ZMnD6QZAE6",
"_exa_published_date": "2024-09-05T00:00:00.000Z"
},
{
"title": "[2406.09246v3] OpenVLA: An Open-Source Vision-Language-Action Model",
"snippet": "[2406.09246v3] OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\n# Title:OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\n> Abstract:Large policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challenges, we introduce OpenVLA, a 7B-parameter open-source VLA trained on a diverse collection of 970k real-world robot demonstrations. OpenVLA builds on a Llama 2 language model combined with a visual encoder that fuses pretrained features from DINOv2 and SigLIP. As a product of the added data diversity and new model components, OpenVLA demonstrates strong results for generalist manipulation, outperforming closed models such as RT-2-X (55B) by 16.5% in absolute task success rate across 29 tasks and multiple robot embodiments, with 7x fewer parameters. We further show that we can effectively fine-tune OpenVLA for new settings, with especially strong generalization results in multi-task environments involving mult",
"source_url": "https://arxiv.org/abs/2406.09246v3",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2406.09246v3",
"_exa_published_date": null
},
{
"title": "Revision History for OpenVLA: An Open-Source... - OpenReview",
"snippet": "Revisions | OpenReview\n\nLoading",
"source_url": "https://openreview.net/revisions?id=ZMnD6QZAE6",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/revisions?id=ZMnD6QZAE6",
"_exa_published_date": null
},
{
"title": "A Vision-Language-Action Model for Affordable and Efficient Robotics",
"snippet": "SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\n# SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\nVision-language models (VLMs) pretrained on large-scale multimodal datasets encode rich visual and linguistic knowledge, making them a strong foundation for robotics. Rather than training robotic policies from scratch, recent approaches adapt VLMs into vision-language-action (VLA) models that enable natural language-driven perception and control. However, existing VLAs are typically massiveoften with billions of parametersleading to high training costs and limited real-world deployability. Moreover, they rely on academic and industrial datasets, overlooking the growing availability of community-collected data from affordable robotic platforms. In this work, we present SmolVLA, a small, efficient, and community-driven VLA that drastically reduces both training and inference costs, while retaining competitive performance. SmolVLA is designed to be trained on a single GPU and deployed on consumer-grade GPUs or even CPUs. To further improve responsiveness, we introduce an asynchronous inference stack decoupling perception and action prediction from action execution, allowing higher control rates with chunked action generation. Despite its compact size, SmolVLA achieves performance comparable to VLAs that are 10 $\\times$ larger. We evaluate SmolVLA on a range of both simulated as well as real-world robotic bench",
"source_url": "https://arxiv.org/abs/2506.01844",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2506.01844",
"_exa_published_date": "2025-06-02T00:00:00.000Z"
},
{
"title": "[PDF] SmolVLA: A Vision-Language-Action Model for Affordable ... - arXiv",
"snippet": "SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\n# SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\nVision-language models (VLMs) pretrained on large-scale multimodal datasets encode rich visual and linguistic knowledge, making them a strong foundation for robotics. Rather than training robotic policies from scratch, recent approaches adapt VLMs into vision-language-action (VLA) models that enable natural language-driven perception and control. However, existing VLAs are typically massiveoften with billions of parametersleading to high training costs and limited real-world deployability. Moreover, they rely on academic and industrial datasets, overlooking the growing availability of community-collected data from affordable robotic platforms. In this work, we present SmolVLA, a small, efficient, and community-driven VLA that drastically reduces both training and inference costs, while retaining competitive performance. SmolVLA is designed to be trained on a single GPU and deployed on consumer-grade GPUs or even CPUs. To further improve responsiveness, we introduce an asynchronous inference stack decoupling perception and action prediction from action execution, allowing higher control rates with chunked action generation. Despite its compact size, SmolVLA achieves performance comparable to VLAs that are 10 $\\times$ larger. We evaluate SmolVLA on a range of both simulated as well as real-world robotic bench",
"source_url": "https://arxiv.org/pdf/2506.01844",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2506.01844",
"_exa_published_date": "2025-06-02T00:00:00.000Z"
},
{
"title": "SmolVLA: A vision-language-action model for affordable and ... - arXiv",
"snippet": "SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\n# SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\nVision-language models (VLMs) pretrained on large-scale multimodal datasets encode rich visual and linguistic knowledge, making them a strong foundation for robotics. Rather than training robotic policies from scratch, recent approaches adapt VLMs into vision-language-action (VLA) models that enable natural language-driven perception and control. However, existing VLAs are typically massiveoften with billions of parametersleading to high training costs and limited real-world deployability. Moreover, they rely on academic and industrial datasets, overlooking the growing availability of community-collected data from affordable robotic platforms. In this work, we present SmolVLA, a small, efficient, and community-driven VLA that drastically reduces both training and inference costs, while retaining competitive performance. SmolVLA is designed to be trained on a single GPU and deployed on consumer-grade GPUs or even CPUs. To further improve responsiveness, we introduce an asynchronous inference stack decoupling perception and action prediction from action execution, allowing higher control rates with chunked action generation. Despite its compact size, SmolVLA achieves performance comparable to VLAs that are 10 $\\times$ larger. We evaluate SmolVLA on a range of both simulated as well as real-world robotic bench",
"source_url": "https://arxiv.org/html/2506.01844v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2506.01844v1",
"_exa_published_date": "2025-06-02T00:00:00.000Z"
},
{
"title": "[PDF] SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics | Semantic Scholar",
"snippet": "[PDF] SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics | Semantic Scholar \n\nNavigate Paper Download (opens in a new tab) Share\n[...]\n```\n@article{Shukor2025SmolVLAAV,\n title={SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},\n author={Mustafa Shukor and Dana Aubakirova and Francesco Capuano and Pepijn Kooijmans and Steven Palma and Adil Zouitine and Michel Aractingi and Caroline Pascal and Martino Russi and Andr{\\'e}s Marafioti and Simon Alibert and Matthieu Cord and Thomas Wolf and R{\\'e}mi Cad{\\`e}ne},\n journal={ArXiv},\n year={2025},\n volume={abs/2506.01844},\n url={https://api.semanticscholar.org/CorpusID:279119427}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/6ab4d113676d00e74b55e918fee4c7affaa8652f",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://www.semanticscholar.org/reader/6ab4d113676d00e74b55e918fee4c7affaa8652f",
"_exa_published_date": "2025-06-02T13:54:14.000Z"
},
{
"title": "Lite VLA: Efficient Vision-Language-Action Control on CPU-Bound Edge Robots",
"snippet": "By leveraging NF4 quantization and the llama-cpp runtime, the proposed LiteVLA implementation pioneers the CPU-only deployment path, achieving functional asynchronous visuomotor control on the low-cost Raspberry Pi 4. This represents a novel deployment strategy not demonstrated by prior GPU-centric VLA frameworks such as SmolVLA by Shukor et al. [22], whose work focused primarily on static robotic arms. Beyond proving technical feasibility, this work establishes a scalable methodology for deploying generalist robot intelligence under strict computational budgets.\n[...]\nParameter-efficient adaptation. We fine-tune a compact SmolVLM backbone using LoRA (rank 8, $\\alpha{=}8$ , dropout 0.1) to specialize visuomotor translation under tight memory/compute budgets (Alg. 1; Sec. III, pp. 23).\n[...]\n. 4).\n[...]\nLarge-scale multimodal systems such as PaLM-E, SayCan, and RT-2 have shown that unified language-conditioned reasoning enables robots to follow natural language commands and execute complex manipulation tasks. However, these approaches rely heavily on cloud-based computation and high-end GPUs, making them impractical for resource-limited or field-deployed robots. SMolVLA by Shukor et al. [22] introduced a small and efficient vision-language-action framework designed for community-driven robotic experimentation. It demonstrated that compact multimodal transformers could achieve competitive visuomotor reasoning performance while running on consumer-grade GPUs or CPUs. Nonetheles",
"source_url": "https://arxiv.org/html/2511.05642",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2511.05642",
"_exa_published_date": null
},
{
"title": "A Vision-Language-Action Flow Model for General Robot Control",
"snippet": "𝜋₀: A Vision-Language-Action Flow Model for General Robot Control\n[...]\n# $\\pi_{0}$ : A Vision-Language-Action Flow Model for General Robot Control\n[...]\nholds tremendous promise to unlock\n[...]\nfull potential of flexible, general, and dexterous\n[...]\nsystems, as well as to address some of\n[...]\ndeepest questions in artificial intelligence\n[...]\nHowever, bringing robot learning to\n[...]\nlevel of generality required for effective real-world systems faces major obstacles in terms of data, generalization, and robustness. In this paper, we discuss how generalist robot policies (i.e., robot foundation models) can address these challenges, and how\n[...]\ncan design effective generalist robot policies for complex and highly dexterous tasks. We propose a novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge. We then discuss how this model can be trained on a large and diverse dataset from multiple dexterous robot platforms, including single-arm robots, dual-arm robots, and mobile manipulators. We evaluate our model in terms of\n[...]\nability to perform tasks via direct prompting, follow language instructions from people and from a high-level VLM policy, and\n[...]\nability to acquire new skills via fine-tuning. Our results cover a wide variety of tasks, such as laundry folding, table cleaning, and assembling boxes.\n[...]\nIn this paper, we present a prototype model and learning framework, which we call $\\pi_",
"source_url": "https://arxiv.org/abs/2410.24164",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2410.24164",
"_exa_published_date": "2024-10-31T00:00:00.000Z"
},
{
"title": "π: A Vision-Language-Action Flow Model for General Robot Control | OpenReview",
"snippet": "π: A Vision-Language-Action Flow Model for General Robot Control | OpenReview\n[...]\n## π: A Vision-Language-Action Flow Model for General Robot Control\n[...]\nAbstract: Robot learning holds tremendous promise to unlock the full potential of flexible, general, and dexterous robot systems, as well as to address some of the deepest questions in artificial intelligence. However, bringing robot learning to the level of generality required for effective real-world systems faces major obstacles in terms of data, generalization, and robustness. In this paper, we discuss how generalist robot policies (i.e., robot foundation models) can address these challenges, and how we can design effective generalist robot policies for complex and highly dexterous tasks. We propose a novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge. We then discuss how this model can be trained on a large and diverse dataset from multiple dexterous robot platforms, including single-arm robots, dual-arm robots, and mobile manipulators. We evaluate our model in terms of its ability to perform tasks in zero shot after pre-training, follow language instructions from people and from a high-level VLM policy, and its ability to acquire new skills via fine-tuning. Our results cover a wide variety of tasks, such as laundry folding, table cleaning, and assembling boxes.",
"source_url": "https://openreview.net/forum?id=38a45ho9Nq",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=38a45ho9Nq",
"_exa_published_date": null
},
{
"title": "Efficient Action Tokenization for Vision-Language-Action Models",
"snippet": "Autoregressive sequence models, such as Transformer-based vision-language action (VLA) policies, can be tremendously effective for capturing complex and generalizable robotic behaviors. However, such models require us to choose a tokenization of our continuous action signals, which determines how the discrete symbols predicted by the model map to continuous robot actions. We find that current approaches for robot action tokenization, based on simple per-dimension, per-timestep binning schemes, typically perform poorly when learning dexterous skills from high-frequency robot data. To address this challenge, we propose a new compression-based tokenization scheme for robot actions, based on the discrete cosine transform. Our tokenization approach, Frequency-space Action Sequence Tokenization (FAST), enables us to train autoregressive VLAs for highly dexterous and high-frequency tasks where standard discretization methods fail completely. Based on FAST, we release FAST+, a universal robot action tokenizer, trained on 1M real robot action trajectories. It can be used as a black-box tokenizer for a wide range of robot action sequences, with diverse action spaces and control frequencies. Finally, we show that, when combined with the $\\bm{\\pi_{0}}$ VLA, our method can scale to training on 10k hours of robot data and match the performance of diffusion VLAs, while reducing training time by up to 5x.\n[...]\nFigure 4: Overview of the FAST action tokenization pipeline. Given a normalized c",
"source_url": "https://arxiv.org/abs/2501.09747",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2501.09747",
"_exa_published_date": "2025-01-16T00:00:00.000Z"
},
{
"title": "FAST: Efficient Action Tokenization for Vision-Language ... - arXiv",
"snippet": "Autoregressive sequence models, such as Transformer-based vision-language action (VLA) policies, can be tremendously effective for capturing complex and generalizable robotic behaviors. However, such models require us to choose a tokenization of our continuous action signals, which determines how the discrete symbols predicted by the model map to continuous robot actions. We find that current approaches for robot action tokenization, based on simple per-dimension, per-timestep binning schemes, typically perform poorly when learning dexterous skills from high-frequency robot data. To address this challenge, we propose a new compression-based tokenization scheme for robot actions, based on the discrete cosine transform. Our tokenization approach, Frequency-space Action Sequence Tokenization (FAST), enables us to train autoregressive VLAs for highly dexterous and high-frequency tasks where standard discretization methods fail completely. Based on FAST, we release FAST+, a universal robot action tokenizer, trained on 1M real robot action trajectories. It can be used as a black-box tokenizer for a wide range of robot action sequences, with diverse action spaces and control frequencies. Finally, we show that, when combined with the $\\bm{\\pi_{0}}$ VLA, our method can scale to training on 10k hours of robot data and match the performance of diffusion VLAs, while reducing training time by up to 5x.\n[...]\nFigure 4: Overview of the FAST action tokenization pipeline. Given a normalized c",
"source_url": "https://arxiv.org/html/2501.09747v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2501.09747v1",
"_exa_published_date": "2025-01-16T00:00:00.000Z"
},
{
"title": "ActionCodec: What Makes for Good Action Tokenizers",
"snippet": "without any robotics\n[...]\nintroduce ActionCodec, a robust action tokenizer that integrates the\n[...]\n. Moreover, ActionCodec leverages Residual Vector Quantization (RVQ) (Lee et al., 2022) post-training to refine reconstruction fidelity and incorporates embodiment-specific soft prompts to facilitate knowledge transfer across diverse robotic platforms\n[...]\nthat VLA\n[...]\nActionCodec, without any additional architectural modifications\n[...]\nefficiency, success rates\n[...]\n. ActionCodec achieves SOTA performance in both\n[...]\nenvironments, providing a systematic\n[...]\nfor the future of VQ-\n[...]\nOur contributions are\n[...]\nas follows:\n[...]\nTokenization Schemes\n[...]\n20\n[...]\nsuffers from low training efficiency and ignores the\n[...]\nparallel decoding (\n[...]\n(Goy\n[...]\n., 2025),\n[...]\nfundamental inefficiencies of heuristic binning. Other\n[...]\nrepresent actions as strings for direct VLM prediction (Hancock et\n[...]\nhowever, this approach\n[...]\nno significant performance benefits while greatly increasing the token budget and extending latency to several seconds, limiting\n[...]\nPertsch et al., 2025) introduces Byte-Pair Encoding (BPE) on frequency-domain signals; however, its reliance on fixed geometric priors limits its capacity for cross-embodiment knowledge transfer. Data-driven schemes, particularly those based on Vector Quantization (VQ) (Wang et al., 2025b; Belkhale and Sadigh, 2024; Mete et al., 2024; Lee et al., 2024), offer a more flexible alternative by learning disc",
"source_url": "https://arxiv.org/abs/2602.15397",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2602.15397",
"_exa_published_date": null
},
{
"title": "OAT: Ordered Action Tokenization",
"snippet": "action tokenization\n[...]\nTo bridge this gap, we introduce Ordered Action Tokenization (OAT), a learned action tokenizer that discretizes continuous action chunks into highly compressed and causally ordered token sequences. OAT employs transformer-based register tokens to aggregate temporal information, finite scalar quantization (FSQ) to construct a discrete bottleneck, and nested dropout to explicitly induce ordering that aligns the latent space with autoregressive generation. The resulting tokenization ensures that any token prefix corresponds to a plausible action chunk. Beyond improved modelability, the ordered structure learned by OAT enables a key capability absent from prior approaches: prefix-based decoding. Autoregressive policies may terminate generation early and still produce valid actions, yielding a natural trade-off between computation and action fidelity. As additional tokens are generated, decoded actions are progressively refined.\n[...]\nAn alternative line of work explores frequency-domain compression, for instance Frequency-space Action Sequence Tokenization (FAST) [49], which employs the Discrete Cosine Transform (DCT) to decompose action chunks into frequency coefficients, followed by Byte Pair Encoding (BPE) [18]. FAST achieves high information density (P.1), and crucially, its low-frequency components first then high-frequency components ordering (P.3) improves downstream autoregressive policies: early token predictions capture the overall trajectory s",
"source_url": "https://arxiv.org/html/2602.04215",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2602.04215",
"_exa_published_date": null
},
{
"title": "PD-VLA: Accelerating Vision-Language-Action Model Integrated with Action Chunking via Parallel Decoding",
"snippet": "the above challenges, we present a novel parallel decoding framework for the mainstream VLA model with action chunking, called Parallel\n[...]\nfor VLA (PD-VLA). Fig. 1 illustrates the\n[...]\nconcept of our parallel decoding approach. Our\n[...]\naction decoding as a system of\n[...]\nsolved through parallel fixed-point iteration methods, e.g., Jacobi fix-point iteration method [32]. This approach preserves\n[...]\nimproving decoding speed\n[...]\nthat we only accelerate the decoding process\n[...]\nVLA inference.\n[...]\n, our method enables friendly\n[...]\ntraining-free acceleration without redesign and modification of models (\n[...]\n, our method\n[...]\nsynergy with existing acceleration\n[...]\nVarious acceleration strategies, including quantization [21] and token pruning [5], have been effectively applied to LLMs, yet they often fail to meet the stringent real-time requirements of action generation. Efforts to enhance efficiency have led to architectural modifications in VLA models, such as DeeR-VLA [43], which dynamically adjusts inference depth, and QAIL [33], which integrates quantization-aware training. Further innovations, like RoboMamba [25] and TinyVLA [41], replace traditional attention mechanisms or focus on developing lightweight models from the ground up, frequently necessitating model re-training and additional data collection. Meanwhile, VLA-Cache [42] selectively caches static tokens and recomputes only dynamic or task-relevant ones. FAST [34] proposes a compression-based toke",
"source_url": "https://arxiv.org/html/2503.02310v2",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2503.02310v2",
"_exa_published_date": null
},
{
"title": "Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware | OpenReview",
"snippet": "Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware | OpenReview\n[...]\n## Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware\n[...]\nAbstract: Fine manipulation tasks, such as threading cable ties or slotting a battery, are notoriously difficult for robots because they require precision, careful coordination of contact forces, and closed-loop visual feedback. Performing these tasks typically requires high-end robots, accurate sensors, or careful calibration, which can be expensive and difficult to set up. Can learning enable low-cost and imprecise hardware to perform these fine manipulation tasks? We present a low-cost system that performs end-to-end imitation learning directly from real demonstrations, collected with a custom teleoperation interface. Imitation learning, however, presents its own challenges, particularly in high-precision domains: errors in the policy can compound over time, and human demonstrations can be non-stationary. To address these challenges, we develop a simple yet novel algorithm, Action Chunking with Transformers (ACT), which learns a generative model over action sequences. ACT allows the robot to learn 6 difficult tasks in the real world, such as opening a translucent condiment cup and slotting a battery with 80-90% success, with only 10 minutes worth of demonstrations.",
"source_url": "https://openreview.net/forum?id=e8Eu1lqLaf",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=e8Eu1lqLaf",
"_exa_published_date": "2023-07-09T06:52:40.000Z"
},
{
"title": "Learning Fine-Grained Bimanual Manipulation with Low-Cost ... - arXiv",
"snippet": "Fine manipulation tasks, such as threading cable ties or slotting a battery, are notoriously difficult for robots because they require precision, careful coordination of contact forces, and closed-loop visual feedback. Performing these tasks typically requires high-end robots, accurate sensors, or careful calibration, which can be expensive and difficult to set up. Can learning enable low-cost and imprecise hardware to perform these fine manipulation tasks? We present a low-cost system that performs end-to-end imitation learning directly from real demonstrations, collected with a custom teleoperation interface. Imitation learning, however, presents its own challenges, particularly in high-precision domains: errors in the policy can compound over time, and human demonstrations can be non-stationary. To address these challenges, we develop a simple yet novel algorithm, Action Chunking with Transformers (ACT), which learns a generative model over action sequences. ACT allows the robot to learn 6 difficult tasks in the real world, such as opening a translucent condiment cup and slotting a battery with 80-90% success, with only 10 minutes worth of demonstrations. Project website: tonyzhaozh.github.io/aloha\n[...]\nImitation learning algorithm. Tasks that require precision and visual feedback present a significant challenge for imitation learning, even with high-quality demonstrations. Small errors in the predicted action can incur large differences in the state, exacerbating the “co",
"source_url": "https://arxiv.org/abs/2304.13705",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2304.13705",
"_exa_published_date": "2023-04-23T00:00:00.000Z"
},
{
"title": "Learning Bimanual Manipulation via Action Chunking and Inter-Arm Coordination with Transformers",
"snippet": "coordinated biman\n[...]\n. To address the\n[...]\narms, particularly for synchronized actions. Therefore, we propose a novel imitation learning architecture that predicts cooperative actions. We differentiate the architecture for both arms and add an intermediate encoder layer, Inter-Arm Coordinated transformer Encoder (IACE),\n[...]\nInter-Arm Coordinated transformer Encoder (IACE), that can adjust the synchronization and timing of potential bimanual movements against the encoders corresponding to each arm. Our overall model\n[...]\na local Transformer encoder for each\n[...]\narm trajectory, the IACE to facilitate learning biman\n[...]\nactions, and a Transformer decoder to\n[...]\nthe action chunk. We compare two types of Transformer decoders: split decoders and single decoders.\n[...]\nWe build our proposed models on the ACT model to design different encoder and decoder structures. In particular, we propose a new design called the inter-arm coordinated transformer Encoder (IACE), which helps synchronize and time the movements of both arms.\n[...]\npropose basic architectures that consist of encoders\n[...]\narm, designed to leverage the biman\n[...]\nfeatures the IACE, allowing the individual robot arms to learn their trajectories while simultaneously considering the state\n[...]\nThe model should focus on the corresponding wrist camera and joint values to determine the appropriate trajectory for each robot arm. Each arm is supported by its local encoder. Global information is also integrated t",
"source_url": "https://arxiv.org/html/2503.13916",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2503.13916",
"_exa_published_date": null
},
{
"title": "ALPHA-𝛼 and Bi-ACT Are All You Need: Importance of Position and Force Information/Control for Imitation Learning of Unimanual and Bimanual Robotic Manipulation with Low-Cost System",
"snippet": "Autonomous manipulation in everyday tasks requires flexible action generation to handle complex, diverse real-world environments, such as objects with varying hardness and softness. Imitation Learning (IL) enables robots to learn complex tasks from expert demonstrations. However, a lot of existing methods rely on position/unilateral control, leaving challenges in tasks that require force information/control, like carefully grasping fragile or varying-hardness objects. As the need for diverse controls increases, there are demand for low-cost bimanual robots that consider various motor inputs. To address these challenges, we introduce Bilateral Control-Based Imitation Learning via Action Chunking with Transformers(Bi-ACT) and”A” ”L”ow-cost ”P”hysical ”Ha”rdware Considering Diverse Motor Control Modes for Research in Everyday Bimanual Robotic Manipulation (ALPHA- $\\alpha$ ). Bi-ACT leverages bilateral control to utilize both position and force information, enhancing the robots adaptability to object characteristics such as hardness, shape, and weight. The concept of ALPHA- $\\alpha$ is affordability, ease of use, repairability, ease of assembly, and diverse control modes (position, velocity, torque), allowing researchers/developers to freely build control systems using ALPHA- $\\alpha$ . In our experiments, we conducted a detailed analysis of Bi-ACT in unimanual manipulation tasks, confirming its superior performance and adaptability compared to Bi-ACT without force control. Base",
"source_url": "https://arxiv.org/html/2411.09942",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2411.09942",
"_exa_published_date": null
},
{
"title": "[PDF] Learning Fine-Grained Bimanual Manipulation with Low-Cost ...",
"snippet": "W7/A/\u0003cs^u\"Nf|5\n[...]\nǨ,~.a(Ц\u0007\u001ePb%\u0019\u001a?3\u001eS~l\u0012j\u0012tf;m3\n[...]\n7h\n[...]\n4\u001e\u0013L\n[...]\n\u000f9\\ \u0001\u001a 'TG15f-ZB34\u0000[\u0006]ӦbQiuK\u0003g4 h⠯ 8@SB\u0002˔.X\u0007IJ)H\u0007IA\u0011Wd\u001a\n[...]\nZ\u0002O[\u0018^jM\u000f\u001dwEdf\u0014hC\u0004\\xͫ\u0010SB\b7\u001aNbܫ#~87MD\n[...]\n\u0001\n[...]\nʬ`y11\u001fB׶fR\u001fT$,Υ;lv\u0015@+%\u0000O;kS6\"\u0002#J\n[...]\nB2+\u0017.\u0004i\u0015C \u0005 \u0003aGT\u001bW\u0015\u000eגRM;\u0004?9$\u0014&\u0013zTɅz=7-ip1S3 +\u001fI\u0013q^\u0018:.3,\"tv\u001c(c2Mf~\u0010'\u001cӝӨBfd&\u0004\u001d\u00042\b C\u001e\u0006SΩTA8M!Ulŏf-\u0003\u0004ݩF30nR]E\b\u0003]\\r\\n5 QuϬZZ|\u0003SD<_(DU<80x)\u0006phT\u0016L\u0006וX#n^+^q8e0\n[...]\nU*cUOB\u0011:\u0013 *. D:\u001biځ\u0015 l:&:Ybxހl̰?\u00149M ^|ʛ1H6|p37I\u0002\u0001Z棍\u0010B!\\A\u0012KdrT5>ƨ^t\u001e\\|m\u0011>poŅ\u0012#5%w[|+\u0016iXfyXޕݔsM\u0014:91b7\u0015=U\n[...]\n\u0012m\u001d! I4\n[...]\nA<ᯟ \u0006\u0018&x⥌;H ,#n\u0013U?xؔd\u0001]8HoJ}\u0007 ܓ =Z\u0005GT?\u0004㘨\u0007\u001a!5[xDi\u001cGc\\\u000f%2c\u001bd6=dSm;\u0004ݏF\u001e\b1T\u0006 \u0012ch\u0007 \u000f\bV^ԥe_P, \u0010h_O\u0012ٱj\u0004J\u001f?\u0012\u0012y\u0017 :{Zs\u001bx\u001d \\_W!\u0006ЬpU!\u001dlmnBE\u0018g\u0012\u0014wɺ8\u0010N\u0011vM`\u001dYD|=ZI8E\u001c~\\Ϫ1K(\u0006\u0005&\u0015\u0010R\u0012_\u001e\u0000F{ \u001bm&u\\\u0005ݶNg\u00114!\u0005Kh.a3o.'2\\󷉏wi3 >)iJsR 3 FBO\u00187\u0019劍o s,*P\u0012\u001e\u0018\"`G\u000f2l$qYO?r\u001e_P{\u0001ܷG\b `6\u001fz.+mt%9#E!$JtJe$8C6ͥ\u001eÙOV]bnvi)u0fHy'\u0010a_,DM&uǒ|s]D%E>\"\"6 $ ACQ\u0003}\u0011 K/bD:ЗG<\u001cy+wj|U\u0000r\u0016Le5GN|\u0010K Yv`\u0000K J0",
"source_url": "https://openreview.net/pdf/4abc35d9793e56c5b73634eaf903e2495311fbcf.pdf",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/pdf/4abc35d9793e56c5b73634eaf903e2495311fbcf.pdf",
"_exa_published_date": null
},
{
"title": "[1512.03385] Deep Residual Learning for Image Recognition - arXiv",
"snippet": "[1512.03385] Deep Residual Learning for Image Recognition\n[...]\n# Deep Residual Learning for Image Recognition\n[...]\nKaiming He Xiangyu Zhang Shaoqing Ren Jian Sun Microsoft Research {kahe, v-xiangz, v-shren, jiansun}@microsoft.com\n[...]\nDeeper neural networks are more difficult to train. We present a residual learning framework to ease the training of networks that are substantially deeper than those used previously. We explicitly reformulate the layers as learning residual functions with reference to the layer inputs, instead of learning unreferenced functions. We provide comprehensive empirical evidence showing that these residual networks are easier to optimize, and can gain accuracy from considerably increased depth. On the ImageNet dataset we evaluate residual nets with a depth of up to 152 layers—8 $\\times$ deeper than VGG nets [41] but still having lower complexity. An ensemble of these residual nets achieves 3.57% error on the ImageNet test set. This result won the 1st place on the ILSVRC 2015 classification task. We also present analysis on CIFAR-10 with 100 and 1000 layers.\n[...]\nreferenced mapping. To\n[...]\nextreme, if\n[...]\nwould be easier\n[...]\nresidual to zero than\n[...]\nnonlinear layers.\n[...]\non ImageNet [36\n[...]\nOn the ImageNet classification dataset [36], we obtain excellent results by extremely deep residual nets. Our 152-layer residual net is the deepest network ever presented on ImageNet, while still having lower complexity than VGG nets [41]. Our ensem",
"source_url": "https://arxiv.org/abs/1512.03385",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/1512.03385",
"_exa_published_date": "2015-12-10T00:00:00.000Z"
},
{
"title": "[1512.03385] Deep Residual Learning for Image Recognition",
"snippet": "[1512.03385] Deep Residual Learning for Image Recognition\n[...]\n# Deep Residual Learning for Image Recognition\n[...]\nKaiming He Xiangyu Zhang Shaoqing Ren Jian Sun Microsoft Research {kahe, v-xiangz, v-shren, jiansun}@microsoft.com\n[...]\nDeeper neural networks are more difficult to train. We present a residual learning framework to ease the training of networks that are substantially deeper than those used previously. We explicitly reformulate the layers as learning residual functions with reference to the layer inputs, instead of learning unreferenced functions. We provide comprehensive empirical evidence showing that these residual networks are easier to optimize, and can gain accuracy from considerably increased depth. On the ImageNet dataset we evaluate residual nets with a depth of up to 152 layers—8 $\\times$ deeper than VGG nets [41] but still having lower complexity. An ensemble of these residual nets achieves 3.57% error on the ImageNet test set. This result won the 1st place on the ILSVRC 2015 classification task. We also present analysis on CIFAR-10 with 100 and 1000 layers.\n[...]\nreferenced mapping. To\n[...]\nextreme, if\n[...]\nwould be easier\n[...]\nresidual to zero than\n[...]\nnonlinear layers.\n[...]\non ImageNet [36\n[...]\nOn the ImageNet classification dataset [36], we obtain excellent results by extremely deep residual nets. Our 152-layer residual net is the deepest network ever presented on ImageNet, while still having lower complexity than VGG nets [41]. Our ensem",
"source_url": "https://arxiv.org/abs/1512.03385v1",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/1512.03385v1",
"_exa_published_date": null
},
{
"title": "[PDF] Deep Residual Learning for Image Recognition - People | MIT CSAIL",
"snippet": "Deep Residual Learning\nfor Image Recognition\nKaiming He, Xiangyu Zhang, Shaoqing Ren, Jian Sun\nwork done at\nMicrosoft Research Asia\n[...]\nResNet @ ILSVRC & COCO 2015 Competitions\n[...]\n1st places in all five main tracks\n[...]\n• ImageNet Classification: “Ultra-deep” 152-layer nets \n• ImageNet Detection: 16% better than 2nd\n[...]\n• ImageNet Localization: 27% better than 2nd\n[...]\n• COCO Detection: 11% better than 2nd\n[...]\n• COCO Segmentation: 12% better than 2nd\n[...]\n*improvements are relative numbers\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nRevolution of Depth\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiang",
"source_url": "https://pdfs.semanticscholar.org/1cea/9b1931b9e87641708fec43d03f2a58f4d2b0.pdf",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://pdfs.semanticscholar.org/1cea/9b1931b9e87641708fec43d03f2a58f4d2b0.pdf",
"_exa_published_date": null
},
{
"title": "Deep Residual Learning for Image Recognition: A Survey - MDPI",
"snippet": "Deep Residual Learning for Image Recognition: A Survey\n[...]\n# Deep Residual Learning for Image Recognition: A Survey\n[...]\nMuhammad Shafiq\n[...]\n1,* and\n\nZhaoquan Gu\n[...]\n2,3,*\n[...]\nCyberspace Institute of Advanced Technology, Guangzhou University, Guangzhou 510006, China\n[...]\nDepartment of New Networks, Peng Cheng Laboratory, Shenzhen 518055, China\n[...]\nDepartment of Computer Science and Technology, Harbin Institute of Technology, Shenzhen 518055, China\n[...]\nAppl. Sci. 2022, 12(18), 8972; https://doi.org/10.3390/app12188972\n[...]\nDeep Residual Networks have recently been shown to significantly improve the performance of neural networks trained on ImageNet, with results beating all previous methods on this dataset by large margins in the image classification task. However, the meaning of these impressive numbers and their implications for future research are not fully understood yet. In this survey, we will try to explain what Deep Residual Networks are, how they achieve their excellent results, and why their successful implementation in practice represents a significant advance over existing techniques. We also discuss some open questions related to residual learning as well as possible applications of Deep Residual Networks beyond ImageNet. Finally, we discuss some issues that still need to be resolved before deep residual learning can be applied on more complex problems.\n[...]\ndeep residual learning for image recognition\n[...]\ndeep residual learning;\n[...]\nDeep resid",
"source_url": "https://www.mdpi.com/2076-3417/12/18/8972",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://www.mdpi.com/2076-3417/12/18/8972",
"_exa_published_date": null
},
{
"title": "[1603.05027] Identity Mappings in Deep Residual Networks - arXiv",
"snippet": "Kaiming He Xiangyu Zhang Shaoqing Ren Jian Sun\n[...]\nDeep residual networks [1] have emerged as a family of extremely deep architectures showing compelling accuracy and nice convergence behaviors. In this paper, we analyze the propagation formulations behind the residual building blocks, which suggest that the forward and backward signals can be directly propagated from one block to any other block, when using identity mappings as the skip connections and after-addition activation. A series of ablation experiments support the importance of these identity mappings. This motivates us to propose a new residual unit, which makes training easier and improves generalization. We report improved results using a 1001-layer ResNet on CIFAR-10 (4.62% error) and CIFAR-100, and a 200-layer ResNet on Image\n[...]\n. Code is available at: https://github.com/KaimingHe/resnet-1k-layers.\n[...]\n- [1] He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR. (2016)",
"source_url": "https://arxiv.org/abs/1603.05027",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/1603.05027",
"_exa_published_date": null
},
{
"title": "[2603.15031] Attention Residuals - arXiv",
"snippet": "Untitled Document\n\n$0$ $5$ $10$ $15$ $20$\n\nUntitled Document\n$0$ $5$ $10$ $15$ $20$\nBETA",
"source_url": "https://arxiv.org/abs/2603.15031",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/2603.15031",
"_exa_published_date": "2026-03-16T00:00:00.000Z"
},
{
"title": "SiameseNorm: Breaking the Barrier to Reconciling Pre/Post-Norm",
"snippet": "In this paper, we propose SiameseNorm, an elegant two\n[...]\nstream residual architecture that unifies\n[...]\nmaintain two residual streams\n[...]\nshared parameters:\n[...]\nthe advantages of\n[...]\nnegligible computational overhead\n[...]\nboosts accuracy from\n[...]\n128\n[...]\n639.6\n[...]\nwhere the product denotes an ordered composition of Jacobians from layer N1N-1 down to i+1i+1. Notably, the term 𝐈\\mathbf{I} corresponds to the skip connection, which preserves an explicit identity gradient path. This allows gradients to flow through the network without explicit attenuation, facilitating the training of large scale models. However, it implicitly allows the representation magnitudes to grow unbounded. As noted previously, Pre-Norm exhibits insufficient effective depth, an issue that likely stems from a structural mismatch: As shown in Figure˜2(a), the main path accumulates residual updates without re-normalization, causing hidden state magnitudes to grow with depth (peri-ln). Consequently, deeper blocks encounter a scaling imbalance: they must influence an increasingly high-magnitude main path while being restricted to normalized, fixed-scale inputs. This growing disparity effectively dilutes the relative contribution of deeper layers, thereby limiting the effective depth of the model.\n[...]\nBy maintaining a clean identity path, Pre-Norm ensures stable gradient propagation. However, this comes at the cost of unbounded magnitude growth. As illustrated in Figure 2(a), while the input ",
"source_url": "https://arxiv.org/html/2602.08064v1",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/html/2602.08064v1",
"_exa_published_date": null
},
{
"title": "[PDF] Attention Residuals - arXiv",
"snippet": "Untitled Document\n\n$0$ $5$ $10$ $15$ $20$\n\nUntitled Document\n$0$ $5$ $10$ $15$ $20$\nBETA",
"source_url": "https://arxiv.org/pdf/2603.15031",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/pdf/2603.15031",
"_exa_published_date": "2026-03-16T00:00:00.000Z"
},
{
"title": "[2409.19606] Hyper-Connections - arXiv",
"snippet": "We present hyper-connections, a simple yet effective method that can serve as an alternative to residual connections. This approach specifically addresses common drawbacks observed in residual connection variants, such as the seesaw effect between gradient vanishing and representation collapse. Theoretically, hyper-connections allow the network to adjust the strength of connections between features at different depths and dynamically rearrange layers. We conduct experiments focusing on the pre-training of large language models, including dense and sparse models, where hyper-connections show significant performance improvements over residual connections. Additional experiments conducted on vision tasks also demonstrate similar improvements. We anticipate that this method will be broadly applicable and beneficial across a wide range of AI problems.\n[...]\nDeep learning has achieved tremendous success across various domains, where residual connections (He et al., 2016) have been instrumental in contemporary neural network architectures, including transformers and CNNs. Residual connections help mitigate the problem of gradient vanishing, enabling the effective training of very deep networks. However, it is important to acknowledge that residual connections are not infallible solutions and still present limitations that remain unresolved.\n[...]\nDriven by the limitations of residual connections, an important question arises: Can neural networks autonomously learn the optimal streng",
"source_url": "https://arxiv.org/abs/2409.19606",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/2409.19606",
"_exa_published_date": "2024-09-29T00:00:00.000Z"
},
{
"title": "Hyper-Connections - OpenReview",
"snippet": "Hyper-Connections | OpenReview\n\n## Hyper-Connections\n\n### Defa Zhu, Hongzhi Huang, Zihao Huang, Yutao Zeng, Yunyao Mao, Banggu Wu, Qiyang Min, Xun Zhou\n\nICLR 2025 Postereveryonesince 04 Oct 2024\">Everyone Revisions BibTeX CC BY 4.0\n\nKeywords: Network Architecture, Residual Connections, LLMs, Pre-training\n\nAbstract: We present hyper-connections, a simple yet effective method that can serve as an alternative to residual connections. This approach specifically addresses common drawbacks observed in residual connection variants, such as the seesaw effect between gradient vanishing and representation collapse. Theoretically, hyper-connections allow the network to adjust the strength of connections between features at different depths and dynamically rearrange layers. We conduct experiments focusing on the pre-training of large language models, including dense and sparse models, where hyper-connections show significant performance improvements over residual connections. Additional experiments conducted on vision tasks also demonstrate similar improvements. We anticipate that this method will be broadly applicable and beneficial across a wide range of AI problems.\n\nPrimary Area: foundation or frontier models, including LLMs\n\nCode Of Ethics: I acknowledge that I and all co-authors of this work have read and commit to adhering to the ICLR Code of Ethics.\n\nSubmission Guidelines: I certify that this submission complies with the submission instructions as described on https://iclr.cc/Con",
"source_url": "https://openreview.net/forum?id=9FqARW7dwB",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://openreview.net/forum?id=9FqARW7dwB",
"_exa_published_date": null
},
{
"title": "Birkhoff-Exact Hyper-Connections: Exact Spectral Stability for Deep Residual Networks | OpenReview",
"snippet": "## Birkhoff-Exact Hyper-Connections: Exact Spectral Stability for Deep Residual Networks\n[...]\nKeywords: doubly stochastic matrices, spectral stability, deep residual networks, Birkhoff-von Neumann theorem, hyper-connections, token mixing, extreme depth training, quantization robustness\n[...]\nTL;DR: We propose BE-HC, which uses the Birkhoff-von Neumann theorem to construct exactly doubly stochastic mixing matrices as convex combinations of permutation matrices, enabling stable training at 1000+ layers where prior methods fail.\n[...]\nAbstract: Learnable information routing in deep networks faces the *depth-stability-efficiency trilemma*: architectures that scale to extreme depths often sacrifice efficiency; efficient approaches lack stability guarantees. Prior work uses iterative Sinkhorn-Knopp normalization to approximate doubly stochastic mixing matrices, but residual errors destabilize training beyond several hundred layers. We propose **Birkhoff-Exact Hyper-Connections (BE-HC)**, which leverages the Birkhoff-von Neumann theorem to construct *exactly* doubly stochastic matrices as convex combinations of permutation matrices. This guarantees spectral radius $\\rho = 1$ exactly—not approximately—enabling stable training at unprecedented depths. **Key results:** (1) *Extreme depth:* BE-HC trains stably at **1000 layers**, achieving 35.71% accuracy where ReZero and other baselines fail to converge. (2) *Long context:* BE-HC handles **8K tokens** on a single V100 GPU (22.56% vali",
"source_url": "https://openreview.net/forum?id=jpIjkN1B1Q",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://openreview.net/forum?id=jpIjkN1B1Q",
"_exa_published_date": "2026-03-02T22:01:32.000Z"
},
{
"title": "Ablate and Rescue: A Causal Analysis of Residual Stream Hyper-Connections",
"snippet": "Multi-stream transformer architectures have recently been proposed as a promising direction for managing representation collapse and the vanishing gradient problem for residual connections, yet their internal mechanisms remain unexplored. In particular, the recently introduced Manifold-Constrained Hyper-Connections (mHC) architecture posits multiple residual streams with constrained interaction, but lacks in-depth mechanistic analysis. We present the first open-source mHC language model (https://huggingface.co/wgpeng/mhc-780m) and analyze the multiple-stream architecture with a suite of representation-level metrics and causal interventions to probe how parallel streams encode and utilize information. Specifically, we introduce a systematic stream ablation-and-rescue framework that enables direct causal comparison of residual streams during inference. Through targeted pairwise interventions and controlled recovery experiments, we distinguish functional redundancy from asymmetric utilization and reveal how information is distributed across streams beyond what is observable from representational similarity alone.\n[...]\nHyper-Connections extend the standard transformer residual architecture by allowing multiple residual streams per layer, dynamically mixed through learned routing matrices (He et al., 2015; Zhu et al., 2025). Manifold-Constrained Hyper-Connections (mHC) further refines this framework by imposing geometric constraints on inter-stream mixing (Xie et al., 2026).\n[...",
"source_url": "https://www.arxiv.org/pdf/2603.14833",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://www.arxiv.org/pdf/2603.14833",
"_exa_published_date": null
},
{
"title": "[1711.05101] Decoupled Weight Decay Regularization - arXiv",
"snippet": "[1711.05\n[...]\npled Weight Decay Regularization\n[...]\nIlya Loshchilov & Frank Hutter University of Freiburg Freiburg, Germany, {ilya,fh}@cs.uni-freiburg.de\n[...]\nL2 regularization and weight decay regularization are equivalent for standard stochastic gradient descent (when rescaled by the learning rate), but as we demonstrate this is not the case for adaptive gradient algorithms, such as Adam. While common implementations of these algorithms employ L2 regularization (often calling it “weight decay” in what may be misleading due to the inequivalence we expose), we propose a simple modification to recover the original formulation of weight decay regularization by decoupling the weight decay from the optimization steps taken w.r.t. the loss function. We provide empirical evidence that our proposed modification (i) decouples the optimal choice of weight decay factor from the setting of the learning rate for both standard SGD and Adam and (ii) substantially improves Adams generalization performance, allowing it to compete with SGD with momentum on image classification datasets (on which it was previously typically outperformed by the latter). Our proposed decoupled weight decay has already been adopted by many researchers, and the community has implemented it in TensorFlow and PyTorch; the complete source code for our experiments is available at https://github.com/loshchil/AdamW-and-SGDW\n[...]\nThe main contribution of this paper is to improve regularization in Adam by decoupling ",
"source_url": "https://arxiv.org/abs/1711.05101",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/abs/1711.05101",
"_exa_published_date": "2017-11-14T00:00:00.000Z"
},
{
"title": "[PDF] Decoupled Weight Decay Regularization - arXiv",
"snippet": "DECOUPLED WEIGHT DECAY REGULARIZATION\n[...]\nIlya Loshchilov & Frank Hutter\n[...]\nL2 regularization and weight decay regularization are equivalent for standard\n[...]\nstochastic gradient descent (when rescaled by the learning rate), but as we demon\u0002strate this is not the case for adaptive gradient algorithms, such as Adam. While\n[...]\nexpose), we propose a simple modification to recover the original formulation of\n[...]\nweight decay regularization by decoupling the weight decay from the optimization\n[...]\nsteps taken w.r.t. the loss function. We provide empirical evidence that our pro\u0002posed modification (i) decouples the optimal choice of weight decay factor from\n[...]\nthe setting of the learning rate for both standard SGD and Adam and (ii) substan\u0002tially improves Adams generalization performance, allowing it to compete with\n[...]\nSGD with momentum on image classification datasets (on which it was previously\n[...]\ntypically outperformed by the latter). Our proposed decoupled weight decay has\n[...]\ncommunity has implemented\n[...]\nit in TensorFlow and PyTorch; the complete source code for our experiments is\n[...]\navailable at https://github.com/loshchil/AdamW-and-SGDW\n[...]\nThe main contribution of this paper is to improve regularization in Adam by decoupling the weight\n[...]\ndecay from the gradient-based update. In a comprehensive analysis, we show that Adam generalizes\n[...]\nsubstantially better with decoupled weight decay than with L2 regularization, achieving 15% relative\n[.",
"source_url": "https://arxiv.org/pdf/1711.05101",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/pdf/1711.05101",
"_exa_published_date": "2019-01-04T00:00:00.000Z"
},
{
"title": "Decoupled Weight Decay Regularization - OpenReview",
"snippet": "Decoupled Weight Decay Regularization | OpenReview\n\n## Decoupled Weight Decay Regularization\n\nICLR 2019 Conference Blind SubmissionReaders: Everyone\n\nAbstract: L$_2$ regularization and weight decay regularization are equivalent for standard stochastic gradient descent (when rescaled by the learning rate), but as we demonstrate this is \\emph{not} the case for adaptive gradient algorithms, such as Adam. While common implementations of these algorithms employ L$_2$ regularization (often calling it ``weight decay'' in what may be misleading due to the inequivalence we expose), we propose a simple modification to recover the original formulation of weight decay regularization by \\emph{decoupling} the weight decay from the optimization steps taken w.r.t. the loss function. We provide empirical evidence that our proposed modification (i) decouples the optimal choice of weight decay factor from the setting of the learning rate for both standard SGD and Adam and (ii) substantially improves Adam's generalization performance, allowing it to compete with SGD with momentum on image classification datasets (on which it was previously typically outperformed by the latter). Our proposed decoupled weight decay has already been adopted by many researchers, and the community has implemented it in TensorFlow and PyTorch; the complete source code for our experiments is available at \\url{https://github.com/loshchil/AdamW-and-SGDW}\n\nKeywords: optimization, regularization, weight decay, Adam\n\nCode: ",
"source_url": "https://openreview.net/forum?id=Bkg6RiCqY7",
"discovered_for": [
"method.training"
],
"_exa_id": "https://openreview.net/forum?id=Bkg6RiCqY7",
"_exa_published_date": "2018-09-27T08:45:35.000Z"
},
{
"title": "PyTorch: An Imperative Style, High-Performance Deep Learning Library",
"snippet": "PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\nDeep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it was designed from first principles to support an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several commonly used benchmarks.",
"source_url": "https://papers.nips.cc/paper/2019/hash/bdbca288fee7f92f2bfa9f7012727740-Abstract.html",
"discovered_for": [
"method.training"
],
"_exa_id": "https://papers.nips.cc/paper/2019/hash/bdbca288fee7f92f2bfa9f7012727740-Abstract.html",
"_exa_published_date": null
},
{
"title": "[PDF] An Imperative Style, High-Performance Deep Learning Library - NIPS",
"snippet": "PyTorch: An Imperative Style, High-Performance Deep Learning Library\n\n| | Adam | Paszke | | Sam | Gross | | | Francisco | Massa | | |\n[...]\nAbstract Deep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it provides an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several common benchmarks.\n[...]\nWith the increased interest in deep learning in recent years, there has been an explosion of machine learning tools. Many popular frameworks such as Caffe [1], CNTK [2], TensorFlow [3], and Theano [4], construct a static dataflow graph that represents the computation and which can then be applied repeatedly to batches of data. This approach provides visibility into the whole comput",
"source_url": "https://papers.neurips.cc/paper/9015-pytorch-an-imperative-style-high-performance-deep-learning-library.pdf",
"discovered_for": [
"method.training"
],
"_exa_id": "https://papers.neurips.cc/paper/9015-pytorch-an-imperative-style-high-performance-deep-learning-library.pdf",
"_exa_published_date": null
},
{
"title": "[1912.01703v1] PyTorch: An Imperative Style, High-Performance Deep Learning Library",
"snippet": "[1912.01703v1] PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\n# Title:PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\n> Abstract:Deep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it provides an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several common benchmarks.",
"source_url": "https://arxiv.org/abs/1912.01703v1",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/abs/1912.01703v1",
"_exa_published_date": null
},
{
"title": "PyTorch: An Imperative Style, High-Performance Deep Learning ...",
"snippet": "[1912.01703] PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\n# PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\nAdam Paszke University of Warsaw adam.paszke@gmail.com Sam Gross Facebook AI Research sgross@fb.com Francisco Massa Facebook AI Research fmassa@fb.com Adam Lerer Facebook AI Research alerer@fb.com James Bradbury Google jekbradbury@gmail.com Gregory Chanan Facebook AI Research gchanan@fb.com Trevor Killeen Self Employed killeent@cs.washington.edu Zeming Lin Facebook AI Research zlin@fb.com Natalia Gimelshein NVIDIA ngimelshein@nvidia.com Luca Antiga Orobix luca.antiga@orobix.com Alban Desmaison Oxford University alban@robots.ox.ac.uk Andreas Köpf Xamla andreas.koepf@xamla.com Edward Yang Facebook AI Research ezyang@fb.com Zach DeVito Facebook AI Research zdevito@cs.stanford.edu Martin Raison Nabla martinraison@gmail.com Alykhan Tejani Twitter atejani@twitter.com Sasank Chilamkurthy Qure.ai sasankchilamkurthy@gmail.com Benoit Steiner Facebook AI Research benoitsteiner@fb.com Lu Fang Facebook lufang@fb.com Junjie Bai Facebook jbai@fb.com Soumith Chintala Facebook AI Research soumith@gmail.com\n[...]\nDeep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it provides an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popu",
"source_url": "https://arxiv.org/abs/1912.01703",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/abs/1912.01703",
"_exa_published_date": "2019-12-03T00:00:00.000Z"
}
],
"n_before_dedup": 90,
"n_after_dedup": 71,
"n_removed": 19
}
+472
View File
@@ -0,0 +1,472 @@
% File: corl_2026.sty
%
% Latex templates for the Conference on Robot Learning (CoRL)
%
% This template is heavily inspired by the NeurIPS, ICML, ICLR and IEEE Transactions latex templates.
% Hence we would like to thank: Roman Garnett and the previous mantainers of the NIPS style, Percy Liang and the previous mantainers of the ICML style, Hugo Larochelle for the ICLR style, and Michael Shell for the IEEE Transactions style.
%
% History:
% 2017/04/16 - First revision by Roberto Calandra (roberto.calandra@berkeley.edu).
% Main changes:
% - The abstract is more compact compared to NeurIPS/ICML
% - References are by default using natbib with squared numbers (e.g., [1])
% - DOI fields from the bibtex are automatically converted to hyperlinks
% to the corresponding page
% - acknowledgments are now a command, and the corresponding subsubsection is
% automatically included only in the final version
% 2017/06/12 - Modified to use corlabbrvnat.bst, which order the reference by order of appearance in the paper
% 2017/06/13 - fixed typo
% 2018/05/09 - Slightly modified for CoRL 2018 by Jun Morimoto (xmorimo@atr.jp)
% 2019/01/28 - Slightly modified for CoRL 2019 by Jun Nakanishi (jnakanis@meijo-u.ac.jp)
% 2020/02/02 - Slightly modified for CoRL 2020 by Cynthia Matuszek (cmat@umbc.edu)
% 2020/08/19 - Added preprint option by Roberto Calandra (rcalandra@fb.com)
% 2021/05/06 - Slightly modified for CoRL 2021 by Gerhard Neumann (gerhard.neumann@kit.edu)
% 2022/03/09 - Slightly modified for CoRL 2022 by Minas Liarokapis (minas.liarokapis@auckland.ac.nz)
% 2022/03/06 - Slightly modified for CoRL 2023 by Marc Toussaint (toussaint@tu-berlin.de)
% 2024/03/26 - Slightly modified for CoRL 2024 by David Held (dheld@andrew.cmu.edu)
% 2026/01/10 - Slightly modified for CoRL 2026 by Yoonchang Sung (yoonchang.sung@ntu.edu.sg)
%
% TODO: nohyperref is not working at the moment
%
\NeedsTeXFormat{LaTeX2e}
% Content to be changed from year to year
\ProvidesPackage{corl_2026}[2026/08/15 CORL2026 submission/preprint/camera-ready style file]
\newcommand{\@conferenceordinal}{10th}
\newcommand{\@conferenceyear}{2026}
\newcommand{\@conferencelocation}{Austin TX, USA}
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Accepted options: [final,preprint,nonatbib,nohyperref]
% Declare the final option, which creates camera-ready copy
\newif\if@conferencefinal\@conferencefinalfalse
\DeclareOption{final}{
\@conferencefinaltrue
}
% Declare the preprint option, which creates a camera-ready copy without the corl footnote
\newif\if@preprinttype\@preprinttypefalse
\DeclareOption{preprint}{
\@preprinttypetrue
}
% The natbib package is loaded by default. Declaring the nonatbib option, does not load natbib in case of package clash (users can pass options to natbib via \PassOptionsToPackage)
\newif\if@natbib\@natbibtrue
\DeclareOption{nonatbib}{
\@natbibfalse
}
% The hyperref package is loaded by default. Declaring the nohyperref option, does not load the hyperref.
\DeclareOption{nohyperref}{%
\gdef\nohyperref{1}
}
% Activate the options
\ProcessOptions\relax
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Required packages:
\RequirePackage{lineno}
\RequirePackage{color}
% Load natbib unless told otherwise
\if@natbib
\RequirePackage[square,numbers]{natbib}
\bibliographystyle{corlabbrvnat}
\fi
% set page geometry
\RequirePackage{hyperref} % hyperlinks
\RequirePackage[verbose=true,letterpaper]{geometry}
\AtBeginDocument{
\newgeometry{
textheight=9in,
textwidth=5.5in,
top=1in,
headheight=12pt,
headsep=25pt,
footskip=30pt
}
\@ifpackageloaded{fullpage}
{\PackageWarning{corl_2026}{fullpage package not allowed! Overwriting formatting.}}
{}
}
\ifdefined\nohyperref\else\ifdefined\hypersetup
\definecolor{mydarkblue}{rgb}{0,0.08,0.45}
\hypersetup{ %
pdftitle={},
pdfauthor={},
pdfsubject={Proceedings of the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear)},
pdfkeywords={},
pdfborder=0 0 0,
pdfpagemode=UseNone,
colorlinks=true,
linkcolor=mydarkblue,
citecolor=mydarkblue,
filecolor=mydarkblue,
urlcolor=mydarkblue,
pdfview=FitH}
\ifdefined\isaccepted \else
\hypersetup{pdfauthor={Anonymous Submission}}
\fi
\fi\fi
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%
% fonts
\renewcommand{\rmdefault}{ptm}
\renewcommand{\sfdefault}{phv}
% Create acknowledgments -- only if the option 'final' is activated
\providecommand{\acknowledgments}{}
\renewcommand{\acknowledgments}[1]{%
\if@conferencefinal%
\subsubsection*{Acknowledgments} #1
\fi
\if@preprinttype%
\subsubsection*{Acknowledgments} #1
\fi
}
% handle tweaks for camera-ready copy vs. submission copy
\if@conferencefinal
\newcommand{\@noticestring}{%
\@conferenceordinal\/ Conference on Robot Learning
(CoRL \@conferenceyear), \@conferencelocation.%
}
\else
\if@preprinttype
\newcommand{\@noticestring}{%
% Nothing here.
}
\else
\newcommand{\@noticestring}{%
Submitted to the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear). Do not distribute.%
}
% line numbers for submission
\linenumbers
% fix incompatibilities between lineno and amsmath, if required, by
% transparently wrapping linenomath environments around amsmath
% environments
\AtBeginDocument{%
\@ifpackageloaded{amsmath}{%
\newcommand*\patchAmsMathEnvironmentForLineno[1]{%
\expandafter\let\csname old#1\expandafter\endcsname\csname #1\endcsname
\expandafter\let\csname oldend#1\expandafter\endcsname\csname end#1\endcsname
\renewenvironment{#1}%
{\linenomath\csname old#1\endcsname}%
{\csname oldend#1\endcsname\endlinenomath}%
}%
\newcommand*\patchBothAmsMathEnvironmentsForLineno[1]{%
\patchAmsMathEnvironmentForLineno{#1}%
\patchAmsMathEnvironmentForLineno{#1*}%
}%
\patchBothAmsMathEnvironmentsForLineno{equation}%
\patchBothAmsMathEnvironmentsForLineno{align}%
\patchBothAmsMathEnvironmentsForLineno{flalign}%
\patchBothAmsMathEnvironmentsForLineno{alignat}%
\patchBothAmsMathEnvironmentsForLineno{gather}%
\patchBothAmsMathEnvironmentsForLineno{multline}%
}{}
}
\fi
\fi
% The DOI will now automatically generate a URL link. Pretty cool!
% + Fix for hyperref and DOI (https://www.tug.org/pipermail/tex-live/2012-August/032161.html)
%-----------------------------
\makeatletter
\providecommand{\doi}[1]{%
\begingroup
\let\bibinfo\@secondoftwo
\urlstyle{rm}%
\href{http://dx.doi.org/#1}{%
doi:\discretionary{}{}{}%
\nolinkurl{#1}%
}%
\endgroup
}
% \makeatother
% %-----------------------------
\widowpenalty=10000
\clubpenalty=10000
\flushbottom
\sloppy
% font sizes with reduced leading
\renewcommand{\normalsize}{%
\@setfontsize\normalsize\@xpt\@xipt
\abovedisplayskip 7\p@ \@plus 2\p@ \@minus 5\p@
\abovedisplayshortskip \z@ \@plus 3\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 4\p@ \@plus 3\p@ \@minus 3\p@
}
\normalsize
\renewcommand{\small}{%
\@setfontsize\small\@ixpt\@xpt
\abovedisplayskip 6\p@ \@plus 1.5\p@ \@minus 4\p@
\abovedisplayshortskip \z@ \@plus 2\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 3\p@ \@plus 2\p@ \@minus 2\p@
}
\renewcommand{\footnotesize}{\@setfontsize\footnotesize\@ixpt\@xpt}
\renewcommand{\scriptsize}{\@setfontsize\scriptsize\@viipt\@viiipt}
\renewcommand{\tiny}{\@setfontsize\tiny\@vipt\@viipt}
\renewcommand{\large}{\@setfontsize\large\@xiipt{14}}
\renewcommand{\Large}{\@setfontsize\Large\@xivpt{16}}
\renewcommand{\LARGE}{\@setfontsize\LARGE\@xviipt{20}}
\renewcommand{\huge}{\@setfontsize\huge\@xxpt{23}}
\renewcommand{\Huge}{\@setfontsize\Huge\@xxvpt{28}}
% sections with less space
\providecommand{\section}{}
\renewcommand{\section}{%
\@startsection{section}{1}{\z@}%
{-2.0ex \@plus -0.5ex \@minus -0.2ex}%
{ 1.5ex \@plus 0.3ex \@minus 0.2ex}%
{\large\bf\raggedright}%
}
\providecommand{\subsection}{}
\renewcommand{\subsection}{%
\@startsection{subsection}{2}{\z@}%
{-1.8ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.8ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\subsubsection}{}
\renewcommand{\subsubsection}{%
\@startsection{subsubsection}{3}{\z@}%
{-1.5ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.5ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\paragraph}{}
\renewcommand{\paragraph}{%
\@startsection{paragraph}{4}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subparagraph}{}
\renewcommand{\subparagraph}{%
\@startsection{subparagraph}{5}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subsubsubsection}{}
\renewcommand{\subsubsubsection}{%
\vskip5pt{\noindent\normalsize\rm\raggedright}%
}
% float placement
\renewcommand{\topfraction }{0.85}
\renewcommand{\bottomfraction }{0.4}
\renewcommand{\textfraction }{0.1}
\renewcommand{\floatpagefraction}{0.7}
\newlength{\@nipsabovecaptionskip}\setlength{\@nipsabovecaptionskip}{7\p@}
\newlength{\@nipsbelowcaptionskip}\setlength{\@nipsbelowcaptionskip}{\z@}
\setlength{\abovecaptionskip}{\@nipsabovecaptionskip}
\setlength{\belowcaptionskip}{\@nipsbelowcaptionskip}
% swap above/belowcaptionskip lengths for tables
\renewenvironment{table}
{\setlength{\abovecaptionskip}{\@nipsbelowcaptionskip}%
\setlength{\belowcaptionskip}{\@nipsabovecaptionskip}%
\@float{table}}
{\end@float}
% footnote formatting
\setlength{\footnotesep }{6.65\p@}
\setlength{\skip\footins}{9\p@ \@plus 4\p@ \@minus 2\p@}
\renewcommand{\footnoterule}{\kern-3\p@ \hrule width 12pc \kern 2.6\p@}
\setcounter{footnote}{0}
% paragraph formatting
\setlength{\parindent}{\z@}
\setlength{\parskip }{5.5\p@}
% list formatting
\setlength{\topsep }{4\p@ \@plus 1\p@ \@minus 2\p@}
\setlength{\partopsep }{1\p@ \@plus 0.5\p@ \@minus 0.5\p@}
\setlength{\itemsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\parsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\leftmargin }{3pc}
\setlength{\leftmargini }{\leftmargin}
\setlength{\leftmarginii }{2em}
\setlength{\leftmarginiii}{1.5em}
\setlength{\leftmarginiv }{1.0em}
\setlength{\leftmarginv }{0.5em}
\def\@listi {\leftmargin\leftmargini}
\def\@listii {\leftmargin\leftmarginii
\labelwidth\leftmarginii
\advance\labelwidth-\labelsep
\topsep 2\p@ \@plus 1\p@ \@minus 0.5\p@
\parsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\itemsep \parsep}
\def\@listiii{\leftmargin\leftmarginiii
\labelwidth\leftmarginiii
\advance\labelwidth-\labelsep
\topsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\parsep \z@
\partopsep 0.5\p@ \@plus 0\p@ \@minus 0.5\p@
\itemsep \topsep}
\def\@listiv {\leftmargin\leftmarginiv
\labelwidth\leftmarginiv
\advance\labelwidth-\labelsep}
\def\@listv {\leftmargin\leftmarginv
\labelwidth\leftmarginv
\advance\labelwidth-\labelsep}
\def\@listvi {\leftmargin\leftmarginvi
\labelwidth\leftmarginvi
\advance\labelwidth-\labelsep}
% create title
\providecommand{\maketitle}{}
\renewcommand{\maketitle}{%
\par
\begingroup
\renewcommand{\thefootnote}{\fnsymbol{footnote}}
% for perfect author name centering
\renewcommand{\@makefnmark}{\hbox to \z@{$^{\@thefnmark}$\hss}}
% The footnote-mark was overlapping the footnote-text,
% added the following to fix this problem (MK)
\long\def\@makefntext##1{%
\parindent 1em\noindent
\hbox to 1.8em{\hss $\m@th ^{\@thefnmark}$}##1
}
\thispagestyle{empty}
\@maketitle
\@thanks
\@notice
\endgroup
\let\maketitle\relax
\let\thanks\relax
}
% rules for title box at top of first page
\newcommand{\@toptitlebar}{
\hrule height 4\p@
\vskip 0.25in
\vskip -\parskip%
}
\newcommand{\@bottomtitlebar}{
\vskip 0.29in
\vskip -\parskip
\hrule height 1\p@
\vskip 0.09in%
}
%% keywords as first class citizens
\def\keywords#1{%
% \ifdefined\isaccepted \else
% \par {\bf Keywords:} #1%
% \fi
% \ifdefined\nohyperref\else\ifdefined\hypersetup
% \hypersetup{pdfkeywords={#1}}
% \fi\fi
\ifdefined\isaccepted \else
\begin{quote}
\textbf{Keywords:} #1%
\end{quote}
\fi
\ifdefined\nohyperref\else\ifdefined\hypersetup
\hypersetup{pdfkeywords={#1}}
\fi\fi
}
% create title (includes both anonymized and non-anonymized versions)
\providecommand{\@maketitle}{}
\renewcommand{\@maketitle}{%
\vbox{%
\hsize\textwidth
\linewidth\hsize
\vskip 0.1in
% \@toptitlebar
\centering
{\LARGE\bf \@title\par}
% \@bottomtitlebar
\if@conferencefinal
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\if@preprinttype
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}
Anonymous Author(s) \\
Affiliation \\
Address \\
\texttt{email} \\
\end{tabular}%
\fi
\fi
\vskip 0.3in \@minus 0.1in
}
}
% add conference notice to bottom of first page
\newcommand{\ftype@noticebox}{8}
\newcommand{\@notice}{%
% give a bit of extra room back to authors on first page
\enlargethispage{2\baselineskip}%
\@float{noticebox}[b]%
\footnotesize\@noticestring%
\end@float%
}
% abstract styling
\renewenvironment{abstract}%
{%
% \vskip 0.075in%
% \centerline%
% {\large\bf Abstract}%
% \vspace{0.5ex}%
\begin{quote}%
\textbf{Abstract:}%
}
{
\par%
\vskip 1ex%
% \ifdefined\keywords
% \textbf{Keywords:}%
% \@keywords%
% \else
% \fi
\end{quote}%
}
\endinput
File diff suppressed because it is too large Load Diff
+118
View File
@@ -0,0 +1,118 @@
\relax
\bibstyle{corlabbrvnat}
\providecommand\hyper@newdestlabel[2]{}
\providecommand\HyField@AuxAddToFields[1]{}
\providecommand\HyField@AuxAddToCoFields[2]{}
\citation{chi2023diffusion}
\citation{ho2020denoising}
\citation{peebles2023scalable}
\citation{lipman2023flow,liu2022flow}
\citation{geng2025mean}
\citation{geng2025improved}
\citation{he2016deep,vaswani2017attention}
\citation{kimi2026attention}
\@writefile{toc}{\contentsline {section}{\numberline {1}Introduction}{1}{section.1}\protected@file@percent }
\newlabel{sec:introduction}{{1}{1}{Introduction}{section.1}{}}
\newlabel{sec:introduction@cref}{{[section][1][]1}{[1][1][]1}{}{}{}}
\citation{zhao2023learning,shukor2025smolvla}
\citation{chi2023diffusion}
\citation{ho2020denoising}
\citation{peebles2023scalable}
\citation{lipman2023flow}
\citation{liu2022flow}
\citation{brohan2022rt,brohan2023rt}
\citation{kim2024openvla}
\citation{shukor2025smolvla,pertsch2025fast}
\citation{zhao2023learning}
\citation{geng2025mean}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {section}{\numberline {2}Related Work}{2}{section.2}\protected@file@percent }
\newlabel{sec:related_work}{{2}{2}{Related Work}{section.2}{}}
\newlabel{sec:related_work@cref}{{[section][2][]2}{[1][2][]2}{}{}{}}
\@writefile{toc}{\contentsline {paragraph}{Diffusion and flow policies for robot action generation.}{2}{section*.1}\protected@file@percent }
\@writefile{toc}{\contentsline {paragraph}{Vision-language-action models and action chunking.}{2}{section*.2}\protected@file@percent }
\@writefile{toc}{\contentsline {paragraph}{Mean-flow training for one-step generation.}{2}{section*.3}\protected@file@percent }
\citation{geng2025improved}
\citation{kimi2026attention}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {paragraph}{Residual aggregation and attention residuals.}{3}{section*.4}\protected@file@percent }
\@writefile{toc}{\contentsline {section}{\numberline {3}Method}{3}{section.3}\protected@file@percent }
\newlabel{sec:method}{{3}{3}{Method}{section.3}{}}
\newlabel{sec:method@cref}{{[section][3][]3}{[1][3][]3}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.1}Policy formulation}{3}{subsection.3.1}\protected@file@percent }
\newlabel{sec:policy_formulation}{{3.1}{3}{Policy formulation}{subsection.3.1}{}}
\newlabel{sec:policy_formulation@cref}{{[subsection][1][3]3.1}{[1][3][]3}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.2}Improved Mean Flow action generation}{3}{subsection.3.2}\protected@file@percent }
\newlabel{sec:imf_action_generation}{{3.2}{3}{Improved Mean Flow action generation}{subsection.3.2}{}}
\newlabel{sec:imf_action_generation@cref}{{[subsection][2][3]3.2}{[1][3][]3}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {1}{\ignorespaces Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image.}}{4}{figure.1}\protected@file@percent }
\newlabel{fig:method_overview}{{1}{4}{Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image}{figure.1}{}}
\newlabel{fig:method_overview@cref}{{[figure][1][]1}{[1][4][]4}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {2}{\ignorespaces Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering.}}{4}{figure.2}\protected@file@percent }
\newlabel{fig:imf_attnres_components}{{2}{4}{Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering}{figure.2}{}}
\newlabel{fig:imf_attnres_components@cref}{{[figure][2][]2}{[1][4][]4}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.3}Attention Residual policy transformer}{4}{subsection.3.3}\protected@file@percent }
\newlabel{sec:attnres_policy}{{3.3}{4}{Attention Residual policy transformer}{subsection.3.3}{}}
\newlabel{sec:attnres_policy@cref}{{[subsection][3][3]3.3}{[1][4][]4}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.4}Inference and control loop}{4}{subsection.3.4}\protected@file@percent }
\newlabel{sec:inference_loop}{{3.4}{4}{Inference and control loop}{subsection.3.4}{}}
\newlabel{sec:inference_loop@cref}{{[subsection][4][3]3.4}{[1][4][]4}{}{}{}}
\@writefile{toc}{\contentsline {section}{\numberline {4}Experiments}{4}{section.4}\protected@file@percent }
\newlabel{sec:experiments}{{4}{4}{Experiments}{section.4}{}}
\newlabel{sec:experiments@cref}{{[section][4][]4}{[1][4][]4}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.1}Setup and metrics}{4}{subsection.4.1}\protected@file@percent }
\newlabel{sec:setup_metrics}{{4.1}{4}{Setup and metrics}{subsection.4.1}{}}
\newlabel{sec:setup_metrics@cref}{{[subsection][1][4]4.1}{[1][4][]4}{}{}{}}
\@writefile{lot}{\contentsline {table}{\numberline {1}{\ignorespaces Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt {max\_reward > 4} count.}}{5}{table.1}\protected@file@percent }
\newlabel{tab:socket_results}{{1}{5}{Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt {max\_reward > 4} count}{table.1}{}}
\newlabel{tab:socket_results@cref}{{[table][1][]1}{[1][5][]5}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {3}{\ignorespaces Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image.}}{5}{figure.3}\protected@file@percent }
\newlabel{fig:socket_reward_latency}{{3}{5}{Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image}{figure.3}{}}
\newlabel{fig:socket_reward_latency@cref}{{[figure][3][]3}{[1][5][]5}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.2}Socket peg insertion}{5}{subsection.4.2}\protected@file@percent }
\newlabel{sec:socket_results}{{4.2}{5}{Socket peg insertion}{subsection.4.2}{}}
\newlabel{sec:socket_results@cref}{{[subsection][2][4]4.2}{[1][5][]5}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.3}Object transfer and ablations}{5}{subsection.4.3}\protected@file@percent }
\newlabel{sec:sim_transfer_results}{{4.3}{5}{Object transfer and ablations}{subsection.4.3}{}}
\newlabel{sec:sim_transfer_results@cref}{{[subsection][3][4]4.3}{[1][5][]5}{}{}{}}
\@writefile{lot}{\contentsline {table}{\numberline {2}{\ignorespaces Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt {max\_reward >= 4} count.}}{6}{table.2}\protected@file@percent }
\newlabel{tab:sim_transfer_results}{{2}{6}{Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt {max\_reward >= 4} count}{table.2}{}}
\newlabel{tab:sim_transfer_results@cref}{{[table][2][]2}{[1][5][]6}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {4}{\ignorespaces Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering.}}{6}{figure.4}\protected@file@percent }
\newlabel{fig:sim_transfer_ablation}{{4}{6}{Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering}{figure.4}{}}
\newlabel{fig:sim_transfer_ablation@cref}{{[figure][4][]4}{[1][6][]6}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.4}Derived speed-quality comparisons}{6}{subsection.4.4}\protected@file@percent }
\newlabel{sec:derived_comparisons}{{4.4}{6}{Derived speed-quality comparisons}{subsection.4.4}{}}
\newlabel{sec:derived_comparisons@cref}{{[subsection][4][4]4.4}{[1][6][]6}{}{}{}}
\@writefile{toc}{\contentsline {section}{\numberline {5}Limitations}{6}{section.5}\protected@file@percent }
\newlabel{sec:limitations}{{5}{6}{Limitations}{section.5}{}}
\newlabel{sec:limitations@cref}{{[section][5][]5}{[1][6][]6}{}{}{}}
\@writefile{lot}{\contentsline {table}{\numberline {3}{\ignorespaces Derived comparisons from the 100-rollout logs.}}{7}{table.3}\protected@file@percent }
\newlabel{tab:derived_comparisons}{{3}{7}{Derived comparisons from the 100-rollout logs}{table.3}{}}
\newlabel{tab:derived_comparisons@cref}{{[table][3][]3}{[1][6][]7}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {section}{\numberline {6}Conclusion}{7}{section.6}\protected@file@percent }
\newlabel{sec:conclusion}{{6}{7}{Conclusion}{section.6}{}}
\newlabel{sec:conclusion@cref}{{[section][6][]6}{[1][7][]7}{}{}{}}
\bibdata{refs}
\bibcite{chi2023diffusion}{{1}{2023}{{Chi et~al.}}{{Chi, Xu, Feng, Cousineau, Du, Burchfiel, Tedrake, and Song}}}
\bibcite{ho2020denoising}{{2}{2020}{{Ho et~al.}}{{Ho, Jain, and Abbeel}}}
\bibcite{peebles2023scalable}{{3}{2023}{{Peebles and Xie}}{{}}}
\bibcite{lipman2023flow}{{4}{2023}{{Lipman et~al.}}{{Lipman, Chen, Ben-Hamu, Nickel, and Le}}}
\bibcite{liu2022flow}{{5}{2022}{{Liu et~al.}}{{Liu, Gong, and Liu}}}
\bibcite{geng2025mean}{{6}{2025{}}{{Geng et~al.}}{{Geng, Deng, Bai, Kolter, and He}}}
\bibcite{geng2025improved}{{7}{2025{}}{{Geng et~al.}}{{Geng, Lu, Wu, Shechtman, Kolter, and He}}}
\bibcite{he2016deep}{{8}{2016}{{He et~al.}}{{He, Zhang, Ren, and Sun}}}
\bibcite{vaswani2017attention}{{9}{2017}{{Vaswani et~al.}}{{Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez, Kaiser, and Polosukhin}}}
\bibcite{kimi2026attention}{{10}{2026}{{Kimi Team}}{{}}}
\bibcite{zhao2023learning}{{11}{2023}{{Zhao et~al.}}{{Zhao, Kumar, Levine, and Finn}}}
\bibcite{shukor2025smolvla}{{12}{2025}{{Shukor et~al.}}{{Shukor, Aubakirova, Capuano, Kooijmans, Palma, Zouitine, Aractingi, Pascal, Russi, Marafioti, et~al.}}}
\bibcite{brohan2022rt}{{13}{2022}{{Brohan et~al.}}{{Brohan, Brown, Carbajal, Chebotar, Dabis, Finn, Gopalakrishnan, Hausman, Herzog, Hsu, et~al.}}}
\bibcite{brohan2023rt}{{14}{2023}{{Brohan et~al.}}{{Brohan, Brown, Carbajal, Chebotar, Chen, Choromanski, Ding, Driess, Dubey, Finn, et~al.}}}
\bibcite{kim2024openvla}{{15}{2024}{{Kim et~al.}}{{Kim, Pertsch, Karamcheti, Xiao, Balakrishna, Nair, Rafailov, Foster, Lam, Sanketi, et~al.}}}
\bibcite{pertsch2025fast}{{16}{2025}{{Pertsch et~al.}}{{Pertsch, Stachowicz, Ichter, Driess, Nair, Vuong, Mees, Finn, and Levine}}}
\def\@LN@column{1}
\gdef \@abspage@last{8}
+111
View File
@@ -0,0 +1,111 @@
\begin{thebibliography}{16}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
\providecommand{\doi}[1]{doi: #1}\else
\providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi
\bibitem[Chi et~al.(2023)Chi, Xu, Feng, Cousineau, Du, Burchfiel, Tedrake, and
Song]{chi2023diffusion}
C.~Chi, Z.~Xu, S.~Feng, E.~Cousineau, Y.~Du, B.~Burchfiel, R.~Tedrake, and
S.~Song.
\newblock Diffusion policy: Visuomotor policy learning via action diffusion.
\newblock \emph{arXiv preprint arXiv:2303.04137}, 2023.
\bibitem[Ho et~al.(2020)Ho, Jain, and Abbeel]{ho2020denoising}
J.~Ho, A.~Jain, and P.~Abbeel.
\newblock Denoising diffusion probabilistic models.
\newblock In \emph{Advances in Neural Information Processing Systems}, 2020.
\bibitem[Peebles and Xie(2023)]{peebles2023scalable}
W.~Peebles and S.~Xie.
\newblock Scalable diffusion models with transformers.
\newblock In \emph{IEEE/CVF International Conference on Computer Vision}, 2023.
\bibitem[Lipman et~al.(2023)Lipman, Chen, Ben-Hamu, Nickel, and
Le]{lipman2023flow}
Y.~Lipman, R.~T.~Q. Chen, H.~Ben-Hamu, M.~Nickel, and M.~Le.
\newblock Flow matching for generative modeling.
\newblock In \emph{International Conference on Learning Representations}, 2023.
\bibitem[Liu et~al.(2022)Liu, Gong, and Liu]{liu2022flow}
X.~Liu, C.~Gong, and Q.~Liu.
\newblock Flow straight and fast: Learning to generate and transfer data with
rectified flow.
\newblock \emph{arXiv preprint arXiv:2209.03003}, 2022.
\bibitem[Geng et~al.(2025{\natexlab{a}})Geng, Deng, Bai, Kolter, and
He]{geng2025mean}
Z.~Geng, M.~Deng, X.~Bai, J.~Z. Kolter, and K.~He.
\newblock Mean flows for one-step generative modeling.
\newblock \emph{arXiv preprint arXiv:2505.13447}, 2025{\natexlab{a}}.
\bibitem[Geng et~al.(2025{\natexlab{b}})Geng, Lu, Wu, Shechtman, Kolter, and
He]{geng2025improved}
Z.~Geng, Y.~Lu, Z.~Wu, E.~Shechtman, J.~Z. Kolter, and K.~He.
\newblock Improved mean flows: On the challenges of fastforward generative
models.
\newblock \emph{arXiv preprint arXiv:2512.02012}, 2025{\natexlab{b}}.
\bibitem[He et~al.(2016)He, Zhang, Ren, and Sun]{he2016deep}
K.~He, X.~Zhang, S.~Ren, and J.~Sun.
\newblock Deep residual learning for image recognition.
\newblock In \emph{IEEE Conference on Computer Vision and Pattern Recognition},
2016.
\bibitem[Vaswani et~al.(2017)Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez,
Kaiser, and Polosukhin]{vaswani2017attention}
A.~Vaswani, N.~Shazeer, N.~Parmar, J.~Uszkoreit, L.~Jones, A.~N. Gomez,
L.~Kaiser, and I.~Polosukhin.
\newblock Attention is all you need.
\newblock In \emph{Advances in Neural Information Processing Systems}, 2017.
\bibitem[{Kimi Team}(2026)]{kimi2026attention}
{Kimi Team}.
\newblock Attention residuals.
\newblock \emph{arXiv preprint arXiv:2603.15031}, 2026.
\bibitem[Zhao et~al.(2023)Zhao, Kumar, Levine, and Finn]{zhao2023learning}
T.~Z. Zhao, V.~Kumar, S.~Levine, and C.~Finn.
\newblock Learning fine-grained bimanual manipulation with low-cost hardware.
\newblock \emph{arXiv preprint arXiv:2304.13705}, 2023.
\bibitem[Shukor et~al.(2025)Shukor, Aubakirova, Capuano, Kooijmans, Palma,
Zouitine, Aractingi, Pascal, Russi, Marafioti, et~al.]{shukor2025smolvla}
M.~Shukor, D.~Aubakirova, F.~Capuano, P.~Kooijmans, S.~Palma, A.~Zouitine,
M.~Aractingi, C.~Pascal, M.~Russi, A.~Marafioti, et~al.
\newblock Smolvla: A vision-language-action model for affordable and efficient
robotics.
\newblock \emph{arXiv preprint arXiv:2506.01844}, 2025.
\bibitem[Brohan et~al.(2022)Brohan, Brown, Carbajal, Chebotar, Dabis, Finn,
Gopalakrishnan, Hausman, Herzog, Hsu, et~al.]{brohan2022rt}
A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, J.~Dabis, C.~Finn,
K.~Gopalakrishnan, K.~Hausman, A.~Herzog, J.~Hsu, et~al.
\newblock Rt-1: Robotics transformer for real-world control at scale.
\newblock \emph{arXiv preprint arXiv:2212.06817}, 2022.
\bibitem[Brohan et~al.(2023)Brohan, Brown, Carbajal, Chebotar, Chen,
Choromanski, Ding, Driess, Dubey, Finn, et~al.]{brohan2023rt}
A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, X.~Chen, K.~Choromanski,
T.~Ding, D.~Driess, A.~Dubey, C.~Finn, et~al.
\newblock Rt-2: Vision-language-action models transfer web knowledge to robotic
control.
\newblock In \emph{Conference on Robot Learning}, 2023.
\bibitem[Kim et~al.(2024)Kim, Pertsch, Karamcheti, Xiao, Balakrishna, Nair,
Rafailov, Foster, Lam, Sanketi, et~al.]{kim2024openvla}
M.~J. Kim, K.~Pertsch, S.~Karamcheti, T.~Xiao, A.~Balakrishna, S.~Nair,
R.~Rafailov, E.~Foster, G.~Lam, P.~Sanketi, et~al.
\newblock Openvla: An open-source vision-language-action model.
\newblock \emph{arXiv preprint arXiv:2406.09246}, 2024.
\bibitem[Pertsch et~al.(2025)Pertsch, Stachowicz, Ichter, Driess, Nair, Vuong,
Mees, Finn, and Levine]{pertsch2025fast}
K.~Pertsch, K.~Stachowicz, B.~Ichter, D.~Driess, S.~Nair, Q.~Vuong, O.~Mees,
C.~Finn, and S.~Levine.
\newblock Fast: Efficient action tokenization for vision-language-action
models.
\newblock \emph{arXiv preprint arXiv:2501.09747}, 2025.
\end{thebibliography}
+46
View File
@@ -0,0 +1,46 @@
This is BibTeX, Version 0.99e (TeX Live 2026/Arch Linux)
Capacity: max_strings=200000, hash_size=200000, hash_prime=170003
The top-level auxiliary file: paper.aux
The style file: corlabbrvnat.bst
Database file #1: refs.bib
You've used 16 entries,
2773 wiz_defined-function locations,
663 strings with 8029 characters,
and the built_in function-call counts, 9470 in all, are:
= -- 701
> -- 960
< -- 2
+ -- 323
- -- 306
* -- 860
:= -- 1568
add.period$ -- 48
call.type$ -- 16
change.case$ -- 146
chr.to.int$ -- 15
cite$ -- 32
duplicate$ -- 342
empty$ -- 555
format.name$ -- 324
if$ -- 1880
int.to.chr$ -- 2
int.to.str$ -- 1
missing$ -- 16
newline$ -- 88
num.names$ -- 64
pop$ -- 317
preamble$ -- 1
purify$ -- 130
quote$ -- 0
skip$ -- 294
stack$ -- 0
substring$ -- 32
swap$ -- 24
text.length$ -- 0
text.prefix$ -- 0
top$ -- 0
type$ -- 176
warning$ -- 0
while$ -- 48
width$ -- 0
write$ -- 199
+205
View File
@@ -0,0 +1,205 @@
# Fdb version 4
["bibtex paper"] 1778845630.39442 "paper.aux" "paper.bbl" "paper" 1778845630.92554 0
"./corlabbrvnat.bst" 1778845629.23107 26694 1972207a84216683d105ea25982a4f25 ""
"./refs.bib" 1778845629.22959 5693 df14ac265e44528611e8ea03e1f31a1e ""
"paper.aux" 1778845630.81834 10211 8e70b0c446cf6b07c68e594ef6fef795 "pdflatex"
(generated)
"paper.bbl"
"paper.blg"
(rewritten before read)
["pdflatex"] 1778845630.43219 "paper.tex" "paper.pdf" "paper" 1778845630.92563 0
"/usr/share/texmf-dist/fonts/enc/dvips/base/8r.enc" 1775415801 4850 80dc9bab7f31fb78a000ccfed0e27cab ""
"/usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-t1.enc" 1775415801 2971 def0b6c1f0b107b3b936def894055589 ""
"/usr/share/texmf-dist/fonts/map/fontname/texfonts.map" 1775415801 3524 cb3e574dea2d1052e39280babc910dc8 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8r.tfm" 1775415801 4712 9ef4d7d106579d4b136e1529e1a4533c ""
"/usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8t.tfm" 1775415801 7040 b2bd27e2bfe6f6948cbc3239cae7444f ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm" 1775415801 4524 6bce29db5bc272ba5f332261583fee9c ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm" 1775415801 6880 f19b8995b61c334d78fc734065f6b4d4 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8r.tfm" 1775415801 4408 25b74d011a4c66b7f212c0cc3c90061b ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm" 1775415801 6672 e3ab9e37e925f3045c9005e6d1473d56 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8r.tfm" 1775415801 4640 532ca3305aad10cc01d769f3f91f1029 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8t.tfm" 1775415801 6944 94c55ad86e6ea2826f78ba2240d50df9 ""
"/usr/share/texmf-dist/fonts/tfm/jknappen/ec/ectt1000.tfm" 1775415801 1536 06717a2b50de47d4087ac0e6cd759455 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm" 1775415801 1004 54797486969f23fa377b128694d548df ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex9.tfm" 1775415801 996 a18840b13b499c08ac2de96a99eda4bc ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam10.tfm" 1775415801 916 f87d7c45f9c908e672703b83b72241a3 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam5.tfm" 1775415801 924 9904cf1d39e9767e7a3622f2a125a565 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam7.tfm" 1775415801 928 2dc8d444221b7a635bb58038579b861a ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm10.tfm" 1775415801 908 2921f8a10601f252058503cc6570e581 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm5.tfm" 1775415801 940 75ac932a52f80982a9f8ea75d03a34cf ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm7.tfm" 1775415801 940 228d6584342e91276bf566bcf9716b83 ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmmi6.tfm" 1775415801 1512 f21f83efb36853c0b70002322c1ab3ad ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmmi9.tfm" 1775415801 1524 d89e2d087a9828407a196f428428ef4a ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmr6.tfm" 1775415801 1300 b62933e007d01cfd073f79b963c01526 ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmr9.tfm" 1775415801 1292 6b21b9c2c7bebb38aa2273f7ca0fb3af ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmsy6.tfm" 1775415801 1116 933a60c408fc0a863a92debe84b2d294 ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmsy9.tfm" 1775415801 1116 25a7bf822c58caf309a702ef79f4afbb ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmex10.pfb" 1775415801 30251 6afa5cb1d0204815a708a080681d4674 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi10.pfb" 1775415801 36299 5f9df58c2139e7edcf37c8fca4bd384d ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi6.pfb" 1775415801 37166 8ab3487cbe3ab49ebce74c29ea2418db ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi7.pfb" 1775415801 36281 c355509802a035cadc5f15869451dcee ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi9.pfb" 1775415801 36094 798f80770b3b148ceedd006d487db67c ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr10.pfb" 1775415801 35752 024fb6c41858982481f6968b5fc26508 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr6.pfb" 1775415801 32734 69e00a6b65cedb993666e42eedb3d48f ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr7.pfb" 1775415801 32762 224316ccc9ad3ca0423a14971cfa7fc1 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr9.pfb" 1775415801 33993 9b89b85fd2d9df0482bd47194d1d3bf3 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy10.pfb" 1775415801 32569 5e5ddc8df908dea60932f3c484a54c0d ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy7.pfb" 1775415801 32716 08e384dc442464e7285e891af9f45947 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy9.pfb" 1775415801 32442 c975af247b6702f7ca0c299af3616b80 ""
"/usr/share/texmf-dist/fonts/type1/public/cm-super/sftt1000.pfb" 1775415801 169201 9ebf99020dde51a5086e186761a34e8f ""
"/usr/share/texmf-dist/fonts/type1/urw/helvetic/uhvr8a.pfb" 1775415801 44648 23115b2a545ebfe2c526c3ca99db8b95 ""
"/usr/share/texmf-dist/fonts/type1/urw/times/utmb8a.pfb" 1775415801 44729 811d6c62865936705a31c797a1d5dada ""
"/usr/share/texmf-dist/fonts/type1/urw/times/utmr8a.pfb" 1775415801 46026 6dab18b61c907687b520c72847215a68 ""
"/usr/share/texmf-dist/fonts/type1/urw/times/utmri8a.pfb" 1775415801 45458 a3faba884469519614ca56ba5f6b1de1 ""
"/usr/share/texmf-dist/fonts/vf/adobe/helvetic/phvr8t.vf" 1775415801 2344 44ff28c9ef2fc97180cd884f900fee71 ""
"/usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf" 1775415801 2340 df9c920cc5688ebbf16a93f45ce7bdd3 ""
"/usr/share/texmf-dist/fonts/vf/adobe/times/ptmr8t.vf" 1775415801 2348 91706c542228501c410c266421fbe30c ""
"/usr/share/texmf-dist/fonts/vf/adobe/times/ptmri8t.vf" 1775415801 2328 6cd7df782b09b29cfc4d93e55b6b9a59 ""
"/usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii" 1775415801 71627 94eb9990bed73c364d7f53f960cc8c5b ""
"/usr/share/texmf-dist/tex/generic/bigintcalc/bigintcalc.sty" 1775415801 40635 c40361e206be584d448876bba8a64a3b ""
"/usr/share/texmf-dist/tex/generic/bitset/bitset.sty" 1775415801 33961 6b5c75130e435b2bfdb9f480a09a39f9 ""
"/usr/share/texmf-dist/tex/generic/gettitlestring/gettitlestring.sty" 1775415801 8371 9d55b8bd010bc717624922fb3477d92e ""
"/usr/share/texmf-dist/tex/generic/iftex/iftex.sty" 1775415801 7984 7dbb9280f03c0a315425f1b4f35d43ee ""
"/usr/share/texmf-dist/tex/generic/iftex/ifvtex.sty" 1775415801 1057 525c2192b5febbd8c1f662c9468335bb ""
"/usr/share/texmf-dist/tex/generic/infwarerr/infwarerr.sty" 1775415801 8356 7bbb2c2373aa810be568c29e333da8ed ""
"/usr/share/texmf-dist/tex/generic/intcalc/intcalc.sty" 1775415801 31769 002a487f55041f8e805cfbf6385ffd97 ""
"/usr/share/texmf-dist/tex/generic/kvdefinekeys/kvdefinekeys.sty" 1775415801 5412 d5a2436094cd7be85769db90f29250a6 ""
"/usr/share/texmf-dist/tex/generic/ltxcmds/ltxcmds.sty" 1775415801 17865 1a9bd36b4f98178fa551aca822290953 ""
"/usr/share/texmf-dist/tex/generic/pdfescape/pdfescape.sty" 1775415801 19007 15924f7228aca6c6d184b115f4baa231 ""
"/usr/share/texmf-dist/tex/generic/pdftexcmds/pdftexcmds.sty" 1775415801 20089 80423eac55aa175305d35b49e04fe23b ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex" 1775415801 1016 1c2b89187d12a2768764b83b4945667c ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorearrows.code.tex" 1775415801 43906 06058dc09064474303f3b5dd62d982c0 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreexternal.code.tex" 1775415801 19324 f4e4c6403dd0f1605fd20ed22fa79dea ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoregraphicstate.code.tex" 1775415801 6038 ccb406740cc3f03bbfb58ad504fe8c27 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreimage.code.tex" 1775415801 6911 f6d4cf5a3fef5cc879d668b810e82868 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorelayers.code.tex" 1775415801 4883 42daaf41e27c3735286e23e48d2d7af9 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreobjects.code.tex" 1775415801 2544 8c06d2a7f0f469616ac9e13db6d2f842 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathconstruct.code.tex" 1775415801 44195 5e390c414de027626ca5e2df888fa68d ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathprocessing.code.tex" 1775415801 17311 e001219836e75b16c4af9a112785f30a ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathusage.code.tex" 1775415801 21302 788a79944eb22192a4929e46963a3067 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepatterns.code.tex" 1775415801 9691 3d42d89522f4650c2f3dc616ca2b925e ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepoints.code.tex" 1775415801 33335 dd1fa4814d4e51f18be97d88bf0da60c ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorequick.code.tex" 1775415801 2965 4c2b1f4e0826925746439038172e5d6f ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorerdf.code.tex" 1775415801 5196 2cc249e0ee7e03da5f5f6589257b1e5b ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorescopes.code.tex" 1775415801 20821 7579108c1e9363e61a0b1584778804aa ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreshade.code.tex" 1775415801 35251 5ff5b5b310c5ac882610e0ccc99095e7 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransformations.code.tex" 1775415801 22012 81b34a0aa8fa1a6158cc6220b00e4f10 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransparency.code.tex" 1775415801 8893 e851de2175338fdf7c17f3e091d94618 ""
"/usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/libraries/tikzlibrarytopaths.code.tex" 1775415801 11518 738408f795261b70ce8dd47459171309 ""
"/usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex" 1775415801 186859 0445d9a41a87648b4723e04765409541 ""
"/usr/share/texmf-dist/tex/generic/pgf/libraries/pgflibraryplothandlers.code.tex" 1775415801 32995 ac577023e12c0e4bd8aa420b2e852d1a ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfint.code.tex" 1775415801 3063 8c415c68a0f3394e45cfeca0b65f6ee6 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex" 1775415801 949 cea70942e7b7eddabfb3186befada2e6 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathcalc.code.tex" 1775415801 13272 7777a64fbd07131a37d276b131c17ee2 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfloat.code.tex" 1775415801 104717 9b2393fbf004a0ce7fa688dbce423848 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.base.code.tex" 1775415801 10165 cec5fa73d49da442e56efc2d605ef154 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.basic.code.tex" 1775415801 28178 41c17713108e0795aac6fef3d275fbca ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.code.tex" 1775415801 9649 85779d3d8d573bfd2cd4137ba8202e60 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.comparison.code.tex" 1775415801 3865 ac538ab80c5cf82b345016e474786549 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.integerarithmetics.code.tex" 1775415801 3177 27d85c44fbfe09ff3b2cf2879e3ea434 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.misc.code.tex" 1775415801 11024 0179538121bc2dba172013a3ef89519f ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.random.code.tex" 1775415801 7889 d0e193914ddc35444510f5b569e26b3d ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.round.code.tex" 1775415801 3379 781797a101f647bab82741a99944a229 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.trigonometric.code.tex" 1775415801 92405 f515f31275db273f97b9d8f52e1b0736 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathparser.code.tex" 1775415801 37733 0fe471ac50324723cf6ab693e5c0916c ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathutil.code.tex" 1775415801 8471 c2883569d03f69e8e1cabfef4999cfd7 ""
"/usr/share/texmf-dist/tex/generic/pgf/modules/pgfmodulematrix.code.tex" 1775415801 21211 1e73ec76bd73964d84197cc3d2685b01 ""
"/usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleplot.code.tex" 1775415801 16218 98503859deba28f16813029fd927ed8e ""
"/usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleshapes.code.tex" 1775415801 44792 c4a5a3feba777682c1d16420f2f01a5b ""
"/usr/share/texmf-dist/tex/generic/pgf/pgf.revision.tex" 1775415801 116 760d50e6a16543bf6edb475635793673 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgf.cfg" 1775415801 926 2963ea0dcf6cc6c0a770b69ec46a477b ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-common-pdf.def" 1775415801 5542 32f75a31ea6c3a7e1148cd6d5e93dbb7 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-pdftex.def" 1775415801 12612 7774ba67bfd72e593c4436c2de6201e3 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex" 1775415801 61355 39904e7552da3800a6838d41440943a5 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex" 1775415801 1896 b8e0ca0ac371d74c0ca05583f6313c91 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex" 1775415801 7778 53c8b5623d80238f6a20aa1df1868e63 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex" 1775415801 24033 d8893a1ec4d1bfa101b172754743d340 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex" 1775415801 39784 414c54e866ebab4b801e2ad81d9b21d8 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeyslibraryfiltered.code.tex" 1775415801 37436 50ba7794827e363eec9ea3467c15c6d7 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex" 1775415801 4385 510565c2f07998c8a0e14f0ec07ff23c ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-common.tex" 1775415801 30029 c49ea8f95207c46731469c614daf4e33 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-latex.def" 1775415801 7067 11553488d1600cac6a0cfca012fca111 ""
"/usr/share/texmf-dist/tex/generic/stringenc/stringenc.sty" 1775415801 21514 b7557edcee22835ef6b03ede1802dad4 ""
"/usr/share/texmf-dist/tex/generic/uniquecounter/uniquecounter.sty" 1775415801 7008 f92eaa0a3872ed622bbf538217cd2ab7 ""
"/usr/share/texmf-dist/tex/latex/amsfonts/amsfonts.sty" 1775415801 5949 3f3fd50a8cc94c3d4cbf4fc66cd3df1c ""
"/usr/share/texmf-dist/tex/latex/amsfonts/amssymb.sty" 1775415801 13829 94730e64147574077f8ecfea9bb69af4 ""
"/usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd" 1775415801 961 6518c6525a34feb5e8250ffa91731cff ""
"/usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd" 1775415801 961 d02606146ba5601b5645f987c92e6193 ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsbsy.sty" 1775415801 2222 27db7d52163edae53881b71ff62e754e ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsgen.sty" 1775415801 4173 1b3e76addfb8afcb47db4811d66e1dc6 ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsmath.sty" 1775415801 88471 b1bb09142edddebd46ba986341b867bd ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsopn.sty" 1775415801 4474 c510a88aa5f51b8c773b50a7ee92befd ""
"/usr/share/texmf-dist/tex/latex/amsmath/amstext.sty" 1775415801 2444 9983e1d0683f102e3b190c64a49313aa ""
"/usr/share/texmf-dist/tex/latex/base/article.cls" 1775415801 20144 b966087dda3b194755eb460d32e2ef75 ""
"/usr/share/texmf-dist/tex/latex/base/fontenc.sty" 1775415801 5275 6f9d359641b36842524cdb97716ab75f ""
"/usr/share/texmf-dist/tex/latex/base/size10.clo" 1775415801 8448 686612a86f0e04f41ea577f5ec7e83d8 ""
"/usr/share/texmf-dist/tex/latex/booktabs/booktabs.sty" 1775415801 6078 f1cb470c9199e7110a27851508ed7a5c ""
"/usr/share/texmf-dist/tex/latex/cleveref/cleveref.sty" 1775415801 329481 7fc6b003158402a4c694bc0a1b729308 ""
"/usr/share/texmf-dist/tex/latex/enumitem/enumitem.sty" 1775415801 52272 63d293bc0d496619edb57585740861a2 ""
"/usr/share/texmf-dist/tex/latex/environ/environ.sty" 1775415801 4378 f429f0da968c278653359293040a8f52 ""
"/usr/share/texmf-dist/tex/latex/epstopdf-pkg/epstopdf-base.sty" 1775415801 13886 d1306dcf79a944f6988e688c1785f9ce ""
"/usr/share/texmf-dist/tex/latex/etoolbox/etoolbox.sty" 1775415801 46885 8953c67ffba03252c6090aa19568b8ba ""
"/usr/share/texmf-dist/tex/latex/geometry/geometry.sty" 1775415801 41601 9cf6c5257b1bc7af01a58859749dd37a ""
"/usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg" 1775415801 1213 620bba36b25224fa9b7e1ccb4ecb76fd ""
"/usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg" 1775415801 1224 978390e9c2234eab29404bc21b268d1e ""
"/usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def" 1775415801 19626 23e2822b9b2b5005f4c549ca98b9334d ""
"/usr/share/texmf-dist/tex/latex/graphics/color.sty" 1775415801 7245 a7e8457a46cda4920df85d975267efb4 ""
"/usr/share/texmf-dist/tex/latex/graphics/graphics.sty" 1775415801 18363 69bb4f5538964bfea50d1e6d89cbe69f ""
"/usr/share/texmf-dist/tex/latex/graphics/graphicx.sty" 1775415801 8118 43b99e52946c33a23f5f43b52d5cc5ec ""
"/usr/share/texmf-dist/tex/latex/graphics/keyval.sty" 1775415801 2671 d9941f4bf4750e9b0603c9a2ec54693b ""
"/usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx" 1775415801 2885 9c645d672ae17285bba324998918efd8 ""
"/usr/share/texmf-dist/tex/latex/graphics/trig.sty" 1775415801 4023 e66acf578d6b564c4670fb57ff336a7a ""
"/usr/share/texmf-dist/tex/latex/grfext/grfext.sty" 1775415801 7133 b94bbacbee6e4fdccdc7f810b2aec370 ""
"/usr/share/texmf-dist/tex/latex/hycolor/hycolor.sty" 1775415801 17914 4c28a13fc3d975e6e81c9bea1d697276 ""
"/usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def" 1775415801 48140 0d317d7fb0c7460a10b7b2713db57305 ""
"/usr/share/texmf-dist/tex/latex/hyperref/hyperref.sty" 1775415801 223349 c7928c099a8656537a829ba316c95536 ""
"/usr/share/texmf-dist/tex/latex/hyperref/nameref.sty" 1775415801 11459 697f11f6c439d25d39d2674b99566af4 ""
"/usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def" 1775415801 14249 b94983bbccc8d5739c16cc91d1fd1c3b ""
"/usr/share/texmf-dist/tex/latex/hyperref/puenc.def" 1775415801 117118 2e3ba580751de5583beacf2e5fee69a9 ""
"/usr/share/texmf-dist/tex/latex/kvoptions/kvoptions.sty" 1775415801 22555 6d8e155cfef6d82c3d5c742fea7c992e ""
"/usr/share/texmf-dist/tex/latex/kvsetkeys/kvsetkeys.sty" 1775415801 13815 760b0c02f691ea230f5359c4e1de23a7 ""
"/usr/share/texmf-dist/tex/latex/l3backend/l3backend-pdftex.def" 1775415801 30662 bfd6e864f4ffc5018b0e2b6260c3181c ""
"/usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg" 1775415801 678 4792914a8f45be57bb98413425e4c7af ""
"/usr/share/texmf-dist/tex/latex/lineno/lineno.sty" 1775415801 155535 edbc920f8c4825a66d507b1de69c9ae0 ""
"/usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def" 1775415801 49656 1c61dbb6f95479ba3d6c0033b53590e2 ""
"/usr/share/texmf-dist/tex/latex/microtype/microtype.cfg" 1775415801 27642 f0cea12315babf4d40608f4eeb5c8458 ""
"/usr/share/texmf-dist/tex/latex/microtype/microtype.sty" 1775415801 102845 043c56602c7d94c8a716319df1af5479 ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg" 1775415801 22906 2122f73c0e7dc828f24240c2422dfe25 ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg" 1775415801 5929 2b35ae0f0fb46984dfffa2bc9d09de5c ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg" 1775415801 5594 992ef5c3f8fd1168bb7101a957333065 ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg" 1775415801 12427 a6802929d6bd2a4ba6d4f5db602da2b4 ""
"/usr/share/texmf-dist/tex/latex/multirow/multirow.sty" 1775415801 6696 886c9f3087d0b973ed2c19aa79cb3023 ""
"/usr/share/texmf-dist/tex/latex/natbib/natbib.sty" 1775415801 45456 1c8843383c0bd05870c45fa0ebea6cc2 ""
"/usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgf.sty" 1775415801 1090 bae35ef70b3168089ef166db3e66f5b2 ""
"/usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgfcore.sty" 1775415801 373 00b204b1d7d095b892ad31a7494b0373 ""
"/usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-0-65.sty" 1775415801 21013 f4ff83d25bb56552493b030f27c075ae ""
"/usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-1-18.sty" 1775415801 989 c49c8ae06d96f8b15869da7428047b1e ""
"/usr/share/texmf-dist/tex/latex/pgf/frontendlayer/tikz.sty" 1775415801 339 c2e180022e3afdb99c7d0ea5ce469b7d ""
"/usr/share/texmf-dist/tex/latex/pgf/math/pgfmath.sty" 1775415801 306 c56a323ca5bf9242f54474ced10fca71 ""
"/usr/share/texmf-dist/tex/latex/pgf/systemlayer/pgfsys.sty" 1775415801 443 8c872229db56122037e86bcda49e14f3 ""
"/usr/share/texmf-dist/tex/latex/pgf/utilities/pgffor.sty" 1775415801 348 ee405e64380c11319f0e249fed57e6c5 ""
"/usr/share/texmf-dist/tex/latex/pgf/utilities/pgfkeys.sty" 1775415801 274 5ae372b7df79135d240456a1c6f2cf9a ""
"/usr/share/texmf-dist/tex/latex/pgf/utilities/pgfrcs.sty" 1775415801 325 f9f16d12354225b7dd52a3321f085955 ""
"/usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd" 1775415801 1483 47067fbe7c3ffed1ede7aaa7b8549d7a ""
"/usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd" 1775415801 774 61d7da1e9f9e74989b196d147e623736 ""
"/usr/share/texmf-dist/tex/latex/refcount/refcount.sty" 1775415801 9878 9e94e8fa600d95f9c7731bb21dfb67a4 ""
"/usr/share/texmf-dist/tex/latex/rerunfilecheck/rerunfilecheck.sty" 1775415801 9684 a33a14b82ce60d6e77cb9be689d79ee6 ""
"/usr/share/texmf-dist/tex/latex/tcolorbox/tcolorbox.sty" 1775415801 108907 22d0f1983935b2026bb960e8244facba ""
"/usr/share/texmf-dist/tex/latex/tools/verbatim.sty" 1775415801 7532 26d26e9d8f2ca784270d5da8ec7d102b ""
"/usr/share/texmf-dist/tex/latex/trimspaces/trimspaces.sty" 1775415801 1380 971a51b00a14503ddf754cab24c3f209 ""
"/usr/share/texmf-dist/tex/latex/url/url.sty" 1775415801 12796 8edb7d69a20b857904dd0ea757c14ec9 ""
"/usr/share/texmf-dist/tex/latex/xcolor/xcolor.sty" 1775415801 55384 b454dec21c2d9f45ec0b793f0995b992 ""
"/usr/share/texmf-dist/web2c/texmf.cnf" 1775415801 43569 fd570f2fa160877d211e859f687312ba ""
"/var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map" 1778740648 5398031 9ff4c9df8bd43dc57ad0660436b2232f ""
"/var/lib/texmf/web2c/pdftex/pdflatex.fmt" 1778740615 2328330 42e60d011c56833d8cf99c1c1a8869a1 ""
"corl_2026.sty" 1778845629.23035 14861 2509e8e2d8ee9fa6e5f4021973730f28 ""
"paper.aux" 1778845630.81834 10211 8e70b0c446cf6b07c68e594ef6fef795 "pdflatex"
"paper.bbl" 1778845630.42933 5101 fe72b057f91937b7d0a5ae07d10029fa "bibtex paper"
"paper.out" 1778845630.81934 2172 1cfee1f9ca74e31f8246b69f1b4736b1 "pdflatex"
"paper.tex" 1778845590.90901 26809 1fa56feaf9da4e8fd970f1b626ec5b45 ""
(generated)
"paper.aux"
"paper.log"
"paper.out"
"paper.pdf"
(rewritten before read)
+356
View File
@@ -0,0 +1,356 @@
PWD /home/droid/project/roboimi/workspace/drafts
INPUT /usr/share/texmf-dist/web2c/texmf.cnf
INPUT /var/lib/texmf/web2c/pdftex/pdflatex.fmt
INPUT paper.tex
OUTPUT paper.log
INPUT /usr/share/texmf-dist/tex/latex/base/article.cls
INPUT /usr/share/texmf-dist/tex/latex/base/article.cls
INPUT /usr/share/texmf-dist/tex/latex/base/size10.clo
INPUT /usr/share/texmf-dist/tex/latex/base/size10.clo
INPUT /usr/share/texmf-dist/tex/latex/base/size10.clo
INPUT ./corl_2026.sty
INPUT corl_2026.sty
INPUT /usr/share/texmf-dist/tex/latex/lineno/lineno.sty
INPUT /usr/share/texmf-dist/tex/latex/lineno/lineno.sty
INPUT /usr/share/texmf-dist/tex/latex/etoolbox/etoolbox.sty
INPUT /usr/share/texmf-dist/tex/latex/etoolbox/etoolbox.sty
INPUT /usr/share/texmf-dist/tex/latex/kvoptions/kvoptions.sty
INPUT /usr/share/texmf-dist/tex/latex/kvoptions/kvoptions.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/keyval.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/keyval.sty
INPUT /usr/share/texmf-dist/tex/generic/ltxcmds/ltxcmds.sty
INPUT /usr/share/texmf-dist/tex/generic/ltxcmds/ltxcmds.sty
INPUT /usr/share/texmf-dist/tex/latex/kvsetkeys/kvsetkeys.sty
INPUT /usr/share/texmf-dist/tex/latex/kvsetkeys/kvsetkeys.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/color.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/color.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/latex/natbib/natbib.sty
INPUT /usr/share/texmf-dist/tex/latex/natbib/natbib.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hyperref.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hyperref.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/iftex.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/iftex.sty
INPUT /usr/share/texmf-dist/tex/generic/kvdefinekeys/kvdefinekeys.sty
INPUT /usr/share/texmf-dist/tex/generic/kvdefinekeys/kvdefinekeys.sty
INPUT /usr/share/texmf-dist/tex/generic/pdfescape/pdfescape.sty
INPUT /usr/share/texmf-dist/tex/generic/pdfescape/pdfescape.sty
INPUT /usr/share/texmf-dist/tex/generic/pdftexcmds/pdftexcmds.sty
INPUT /usr/share/texmf-dist/tex/generic/pdftexcmds/pdftexcmds.sty
INPUT /usr/share/texmf-dist/tex/generic/infwarerr/infwarerr.sty
INPUT /usr/share/texmf-dist/tex/generic/infwarerr/infwarerr.sty
INPUT /usr/share/texmf-dist/tex/latex/hycolor/hycolor.sty
INPUT /usr/share/texmf-dist/tex/latex/hycolor/hycolor.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/nameref.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/nameref.sty
INPUT /usr/share/texmf-dist/tex/latex/refcount/refcount.sty
INPUT /usr/share/texmf-dist/tex/latex/refcount/refcount.sty
INPUT /usr/share/texmf-dist/tex/generic/gettitlestring/gettitlestring.sty
INPUT /usr/share/texmf-dist/tex/generic/gettitlestring/gettitlestring.sty
INPUT /usr/share/texmf-dist/tex/generic/stringenc/stringenc.sty
INPUT /usr/share/texmf-dist/tex/generic/stringenc/stringenc.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def
INPUT /usr/share/texmf-dist/tex/generic/intcalc/intcalc.sty
INPUT /usr/share/texmf-dist/tex/generic/intcalc/intcalc.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/puenc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/puenc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/puenc.def
INPUT /usr/share/texmf-dist/tex/latex/url/url.sty
INPUT /usr/share/texmf-dist/tex/latex/url/url.sty
INPUT /usr/share/texmf-dist/tex/generic/bitset/bitset.sty
INPUT /usr/share/texmf-dist/tex/generic/bitset/bitset.sty
INPUT /usr/share/texmf-dist/tex/generic/bigintcalc/bigintcalc.sty
INPUT /usr/share/texmf-dist/tex/generic/bigintcalc/bigintcalc.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def
INPUT /usr/share/texmf-dist/tex/latex/rerunfilecheck/rerunfilecheck.sty
INPUT /usr/share/texmf-dist/tex/latex/rerunfilecheck/rerunfilecheck.sty
INPUT /usr/share/texmf-dist/tex/generic/uniquecounter/uniquecounter.sty
INPUT /usr/share/texmf-dist/tex/generic/uniquecounter/uniquecounter.sty
INPUT /usr/share/texmf-dist/tex/latex/geometry/geometry.sty
INPUT /usr/share/texmf-dist/tex/latex/geometry/geometry.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/ifvtex.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/ifvtex.sty
INPUT /usr/share/texmf-dist/tex/latex/base/fontenc.sty
INPUT /usr/share/texmf-dist/tex/latex/base/fontenc.sty
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd
INPUT /usr/share/texmf-dist/fonts/map/fontname/texfonts.map
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.sty
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.sty
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.cfg
INPUT /usr/share/texmf-dist/tex/latex/booktabs/booktabs.sty
INPUT /usr/share/texmf-dist/tex/latex/booktabs/booktabs.sty
INPUT /usr/share/texmf-dist/tex/latex/multirow/multirow.sty
INPUT /usr/share/texmf-dist/tex/latex/multirow/multirow.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphicx.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphicx.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphics.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphics.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/trig.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/trig.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsmath.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsmath.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsopn.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amstext.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amstext.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsgen.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsgen.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsbsy.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsbsy.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsopn.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amssymb.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amssymb.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amsfonts.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amsfonts.sty
INPUT /usr/share/texmf-dist/tex/latex/enumitem/enumitem.sty
INPUT /usr/share/texmf-dist/tex/latex/enumitem/enumitem.sty
INPUT /usr/share/texmf-dist/tex/latex/tcolorbox/tcolorbox.sty
INPUT /usr/share/texmf-dist/tex/latex/tcolorbox/tcolorbox.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/frontendlayer/tikz.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/frontendlayer/tikz.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgf.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgf.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfrcs.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfrcs.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-common.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-latex.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/pgf.revision.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/pgf.revision.tex
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgfcore.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgfcore.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/systemlayer/pgfsys.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/systemlayer/pgfsys.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeyslibraryfiltered.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgf.cfg
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-pdftex.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-pdftex.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-common-pdf.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex
INPUT /usr/share/texmf-dist/tex/latex/xcolor/xcolor.sty
INPUT /usr/share/texmf-dist/tex/latex/xcolor/xcolor.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathutil.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathparser.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.basic.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.trigonometric.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.random.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.comparison.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.base.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.round.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.misc.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.integerarithmetics.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathcalc.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfloat.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfint.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepoints.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathconstruct.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathusage.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorescopes.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoregraphicstate.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransformations.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorequick.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreobjects.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathprocessing.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorearrows.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreshade.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreimage.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreexternal.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorelayers.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransparency.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepatterns.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorerdf.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleshapes.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleplot.code.tex
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-0-65.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-0-65.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-1-18.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-1-18.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgffor.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgffor.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfkeys.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfkeys.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/latex/pgf/math/pgfmath.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/math/pgfmath.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/libraries/pgflibraryplothandlers.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/libraries/pgflibraryplothandlers.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/modules/pgfmodulematrix.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/libraries/tikzlibrarytopaths.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/libraries/tikzlibrarytopaths.code.tex
INPUT /usr/share/texmf-dist/tex/latex/tools/verbatim.sty
INPUT /usr/share/texmf-dist/tex/latex/tools/verbatim.sty
INPUT /usr/share/texmf-dist/tex/latex/environ/environ.sty
INPUT /usr/share/texmf-dist/tex/latex/environ/environ.sty
INPUT /usr/share/texmf-dist/tex/latex/trimspaces/trimspaces.sty
INPUT /usr/share/texmf-dist/tex/latex/trimspaces/trimspaces.sty
INPUT /usr/share/texmf-dist/tex/latex/cleveref/cleveref.sty
INPUT /usr/share/texmf-dist/tex/latex/cleveref/cleveref.sty
INPUT /usr/share/texmf-dist/tex/latex/l3backend/l3backend-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/l3backend/l3backend-pdftex.def
INPUT ./paper.aux
INPUT ./paper.aux
INPUT paper.aux
OUTPUT paper.aux
INPUT /usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii
INPUT /usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii
INPUT /usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii
INPUT /usr/share/texmf-dist/tex/latex/epstopdf-pkg/epstopdf-base.sty
INPUT /usr/share/texmf-dist/tex/latex/epstopdf-pkg/epstopdf-base.sty
INPUT /usr/share/texmf-dist/tex/latex/grfext/grfext.sty
INPUT /usr/share/texmf-dist/tex/latex/grfext/grfext.sty
INPUT /usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg
INPUT /usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg
INPUT /usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg
INPUT ./paper.out
INPUT ./paper.out
INPUT paper.out
INPUT paper.out
OUTPUT paper.pdf
INPUT ./paper.out
INPUT ./paper.out
OUTPUT paper.out
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam10.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam5.tfm
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm10.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm5.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/jknappen/ec/ectt1000.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT /var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map
INPUT /usr/share/texmf-dist/fonts/enc/dvips/base/8r.enc
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmr8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8r.tfm
INPUT /usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-t1.enc
INPUT /usr/share/texmf-dist/fonts/vf/adobe/helvetic/phvr8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8r.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmr8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8r.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmr9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmr6.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmmi9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmmi6.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmsy9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmsy6.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam10.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm10.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT ./paper.bbl
INPUT ./paper.bbl
INPUT paper.bbl
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8t.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmri8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8r.tfm
INPUT paper.aux
INPUT ./paper.out
INPUT ./paper.out
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmex10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi6.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi7.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi9.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr6.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr7.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr9.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy7.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy9.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/cm-super/sftt1000.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/helvetic/uhvr8a.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/times/utmb8a.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/times/utmr8a.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/times/utmri8a.pfb
+14
View File
@@ -0,0 +1,14 @@
\BOOKMARK [1][-]{section.1}{\376\377\000I\000n\000t\000r\000o\000d\000u\000c\000t\000i\000o\000n}{}% 1
\BOOKMARK [1][-]{section.2}{\376\377\000R\000e\000l\000a\000t\000e\000d\000\040\000W\000o\000r\000k}{}% 2
\BOOKMARK [1][-]{section.3}{\376\377\000M\000e\000t\000h\000o\000d}{}% 3
\BOOKMARK [2][-]{subsection.3.1}{\376\377\000P\000o\000l\000i\000c\000y\000\040\000f\000o\000r\000m\000u\000l\000a\000t\000i\000o\000n}{section.3}% 4
\BOOKMARK [2][-]{subsection.3.2}{\376\377\000I\000m\000p\000r\000o\000v\000e\000d\000\040\000M\000e\000a\000n\000\040\000F\000l\000o\000w\000\040\000a\000c\000t\000i\000o\000n\000\040\000g\000e\000n\000e\000r\000a\000t\000i\000o\000n}{section.3}% 5
\BOOKMARK [2][-]{subsection.3.3}{\376\377\000A\000t\000t\000e\000n\000t\000i\000o\000n\000\040\000R\000e\000s\000i\000d\000u\000a\000l\000\040\000p\000o\000l\000i\000c\000y\000\040\000t\000r\000a\000n\000s\000f\000o\000r\000m\000e\000r}{section.3}% 6
\BOOKMARK [2][-]{subsection.3.4}{\376\377\000I\000n\000f\000e\000r\000e\000n\000c\000e\000\040\000a\000n\000d\000\040\000c\000o\000n\000t\000r\000o\000l\000\040\000l\000o\000o\000p}{section.3}% 7
\BOOKMARK [1][-]{section.4}{\376\377\000E\000x\000p\000e\000r\000i\000m\000e\000n\000t\000s}{}% 8
\BOOKMARK [2][-]{subsection.4.1}{\376\377\000S\000e\000t\000u\000p\000\040\000a\000n\000d\000\040\000m\000e\000t\000r\000i\000c\000s}{section.4}% 9
\BOOKMARK [2][-]{subsection.4.2}{\376\377\000S\000o\000c\000k\000e\000t\000\040\000p\000e\000g\000\040\000i\000n\000s\000e\000r\000t\000i\000o\000n}{section.4}% 10
\BOOKMARK [2][-]{subsection.4.3}{\376\377\000O\000b\000j\000e\000c\000t\000\040\000t\000r\000a\000n\000s\000f\000e\000r\000\040\000a\000n\000d\000\040\000a\000b\000l\000a\000t\000i\000o\000n\000s}{section.4}% 11
\BOOKMARK [2][-]{subsection.4.4}{\376\377\000D\000e\000r\000i\000v\000e\000d\000\040\000s\000p\000e\000e\000d\000-\000q\000u\000a\000l\000i\000t\000y\000\040\000c\000o\000m\000p\000a\000r\000i\000s\000o\000n\000s}{section.4}% 12
\BOOKMARK [1][-]{section.5}{\376\377\000L\000i\000m\000i\000t\000a\000t\000i\000o\000n\000s}{}% 13
\BOOKMARK [1][-]{section.6}{\376\377\000C\000o\000n\000c\000l\000u\000s\000i\000o\000n}{}% 14
Binary file not shown.
+306
View File
@@ -0,0 +1,306 @@
\documentclass{article}
\usepackage{corl_2026}
\usepackage[T1]{fontenc}
\usepackage{microtype}
\usepackage{booktabs}
\usepackage{multirow}
\usepackage{graphicx}
\usepackage{amsmath,amssymb}
\usepackage{enumitem}
\usepackage{tcolorbox}
\usepackage[capitalize]{cleveref}
\usepackage{url}
\title{iMF-AttnRes: Fast Mean-Flow Action Generation with Attention Residuals for Simulated Robot Manipulation}
\author{Anonymous Authors}
\begin{document}
\maketitle
\begin{abstract}
Diffusion-style vision-language-action policies provide expressive action distributions, but iterative inference can be too slow for closed-loop manipulation. We study whether improved Mean Flow (iMF), originally motivated by fast-forward generative modeling, can be adapted to action generation so that a policy uses one to a few flow evaluations rather than long denoising chains. We combine iMF with Attention Residuals (AttnRes), which replace fixed residual accumulation in a deep policy transformer with learned depth-wise aggregation over previous layer outputs. In RoboIMI socket peg insertion, the best iMF-AttnRes policy reaches an average reward of 1513.56 and median reward of 1901.5 with 15.122 ms average inference time, compared with diffusion-style infer100 baselines at 861.53/338.291 ms and 1019.39/397.912 ms. In RoboIMI object transfer, the best iMF-AttnRes policy reaches average reward 526.22 and 44/100 success-like episodes, compared with a native diffusion-policy baseline at 319.2 and 29/100. These results suggest that average-flow action generation is a promising route to low-latency robot policies, while also revealing sensitivity to horizon choices, vision-token design, and simulation-to-real validation.
\end{abstract}
\keywords{Robot learning, imitation learning, flow matching, vision-language-action models}
\section{Introduction}
\label{sec:introduction}
Learning-based robot manipulation policies increasingly use expressive generative action models. Diffusion Policy showed that action diffusion can be an effective visuomotor policy class for manipulation~\citep{black2024pi0,chi2023diffusion}, inheriting the representational flexibility of denoising diffusion models~\citep{ho2020denoising} and transformer-based diffusion backbones~\citep{peebles2023scalable}. However, this expressivity often comes with a deployment cost: the policy must be evaluated repeatedly during sampling. In the RoboIMI socket peg experiments studied here, two diffusion-style VLA baselines use infer100 sampling and require 338.291 ms and 397.912 ms per inference on their recorded rollouts. Such latencies are undesirable when the robot must repeatedly close the perception-action loop.
This paper asks a direct question: can a robot action generator retain the quality benefits of generative policies while reducing inference to one or a few evaluations? We explore this question by adapting improved Mean Flow (iMF) to VLA-style imitation learning. Flow matching and rectified flow formulate generation through continuous vector fields~\citep{lipman2023flow,liu2022flow}. Mean Flow reframes generation around average velocity, enabling one-step generation~\citep{geng2025mean}; iMF further addresses instability caused by substituting conditional velocities into nonlinear mean-flow relations~\citep{geng2025improved}. For robot control, this distinction matters because fewer evaluations directly increase control responsiveness.
We pair iMF with Attention Residuals (AttnRes). Residual connections are essential for deep visual and transformer networks~\citep{he2016deep,vaswani2017attention}, but standard residual accumulation adds all layer outputs with fixed unit weight. The motivation in our notes is to rewrite residual networks as a sum over layer increments, then replace the uniform sum with a learned softmax aggregation. AttnRes implements this view by letting each layer attend over preceding residual outputs~\citep{kimi2026attention}. We use this mechanism in the policy transformer to reduce depth-wise feature dilution while keeping the action generator fast.
Our contributions are:
\begin{enumerate}[leftmargin=*]
\item We formulate an iMF-AttnRes VLA policy for simulated robot imitation learning, combining one-to-few-step average-flow action generation with learned residual aggregation.
\item We provide 100-rollout evaluations on two RoboIMI manipulation environments, socket peg insertion and object transfer, comparing against diffusion-style VLA baselines, a native Diffusion Policy baseline, ACT-style action chunking, and SmolVLA-style compact VLA inference~\citep{zhao2023learning,shukor2025smolvla}.
\item We report both task quality and deployment speed. On socket insertion, the strongest iMF-AttnRes run improves average reward over the ph16 diffusion-style baseline by 1.76$\times$ and inference FPS by 22.97$\times$; on object transfer, the strongest iMF-AttnRes run improves average reward over the native diffusion-policy baseline by 1.65$\times$.
\end{enumerate}
We deliberately keep the claims scoped to simulation: hardware varies across runs, and real-robot transfer remains future work.
\section{Related Work}
\label{sec:related_work}
\paragraph{Diffusion and flow policies for robot action generation.}
Diffusion Policy introduced action diffusion as a visuomotor policy learning framework~\citep{black2024pi0,chi2023diffusion}, building on denoising diffusion objectives~\citep{ho2020denoising}. Diffusion transformers further show that transformer backbones can scale generative modeling~\citep{peebles2023scalable}. These models are expressive because they iteratively refine samples, but iterative refinement is also a latency source. Flow matching provides an alternative continuous generative formulation by learning vector fields that transport probability paths~\citep{lipman2023flow}. Rectified flow emphasizes straighter transport paths and faster sampling~\citep{liu2022flow}. Our method follows this fast-generation lineage but uses the improved mean-flow relation to target one-to-few-step action generation.
\paragraph{Vision-language-action models and action chunking.}
Transformer robot policies such as RT-1 and RT-2 show how large sequence models can map observations and task context to robot actions~\citep{brohan2022rt,brohan2023rt}. OpenVLA studies open VLA modeling at scale~\citep{kim2024openvla}, while compact or efficient VLA models such as SmolVLA and FAST reduce deployment cost through smaller backbones or efficient action tokenization~\citep{shukor2025smolvla,pertsch2025fast}. Action Chunking with Transformers (ACT) predicts chunks of future actions to improve temporal consistency and execution efficiency~\citep{zhao2023learning}. The iMF-AttnRes policy is complementary: it retains chunked execution but changes the generative action sampler so that each chunk can be produced with one to a few evaluations.
\paragraph{Residual aggregation and attention residuals.}
Standard residual networks update $x_t=x_{t-1}+y_t$, so recursively $x_t=y_0+y_1+\cdots+y_t$. Equivalently, the next layer receives a uniform aggregation of all previous residual increments. A natural generalization is
\begin{equation}
y_{t+1}=f_{t+1}\left(\sum_{s=0}^t a_{t+1,s}y_s\right),\qquad a_{t+1,s}\ge0,\quad \sum_{s=0}^t a_{t+1,s}=1.
\end{equation}
AttnRes instantiates the weights with an attention distribution over previous residual outputs,
\begin{equation}
a_{t+1,s}\propto\exp\left(w_{t+1}^{\top}\mathrm{RMSNorm}(y_s)\right).
\end{equation}
This preserves the residual pathway while allowing each layer to select useful earlier representations instead of uniformly accumulating all of them~\citep{kimi2026attention}. We use this mechanism in the policy transformer; our ablations suggest that applying AttnRes too broadly inside the vision stack is not automatically beneficial.
\section{Method}
\label{sec:method}
\subsection{Policy formulation}
\label{sec:policy_formulation}
We consider simulated imitation learning in RoboIMI. At each control step, the policy receives visual observations, robot state, and a task context, and predicts an action chunk of length \texttt{exec}. The main iMF-AttnRes socket and object-transfer policies use ph32/exec16 or related horizon-execution settings. The rollout evaluator records cumulative reward, median reward, mean maximum reward, nonzero-reward episode count, task-specific success-like threshold count, inference FPS, control FPS, and inference latency.
The policy contains three conceptual parts. First, an observation encoder maps visual and state inputs into policy tokens. Second, a transformer policy core uses AttnRes in place of fixed residual accumulation. Third, the action generator uses iMF to predict an average flow over action noise-to-data paths, enabling one-to-few inference evaluations. In the strongest socket variants, the generator uses infer2 or infer3; in the strongest object-transfer run, it uses infer1.
\subsection{Improved Mean Flow action generation}
\label{sec:imf_action_generation}
Classical flow matching trains an instantaneous velocity field. Let $x_t$ denote a sample at time $t$ and let the instantaneous flow satisfy
\begin{equation}
\frac{d x_t}{dt} = v(x_t,t).
\end{equation}
If a policy takes very few integration or denoising steps, using only this local instantaneous velocity can cause large path errors when the learned trajectory is curved. Mean Flow addresses this by training an average velocity over an interval. For two times $r<t$, define
\begin{equation}
\bar{v}(z_t,r,t) = \frac{1}{t-r}\int_r^t v(z_\tau,\tau)d\tau.
\end{equation}
The model receives $(z_t,r,t)$ and predicts $u_\theta(z_t,r,t)$, an approximation to $\bar{v}(z_t,r,t)$. At inference, this average velocity is used to move directly across a larger interval, replacing a long sequence of small denoising or ODE steps.
The useful training identity comes from differentiating $(t-r)\bar{v}$ with respect to $t$:
\begin{equation}
\bar{v}(z_t,r,t)=v(z_t,t) - (t-r)\frac{d}{dt}\bar{v}(z_t,r,t).
\end{equation}
The total derivative contains the dependence of $\bar{v}$ on the current state and time, and can be written as a Jacobian-vector product,
\begin{equation}
\frac{d}{dt}\bar{v}(z_t,r,t)
= \mathrm{jvp}\big(\bar{v},(z,r,t),(v,0,1)\big).
\end{equation}
Mean Flow uses this relation to train an average-velocity predictor for one-step generation~\citep{geng2025mean}. In our setting, the target action sample and noise sample provide a conditional straight-path velocity, but the nonlinear JVP expression should conceptually depend on the marginal velocity field. Directly substituting the conditional velocity inside the nonlinear term can destabilize training. We therefore use the improved Mean Flow correction~\citep{geng2025improved}:
\begin{equation}
V_\theta(z_t,r,t) = u_\theta(z_t,r,t) + (t-r)\mathrm{JVP}_{\mathrm{sg}}\big(u_\theta; v_\theta\big).
\end{equation}
Here $v_\theta$ is obtained from the same network at the instantaneous case, and the stop-gradient JVP direction prevents the nonlinear term from destabilizing the target. The supervised target remains the conditional straight-path velocity available from the imitation action sample and noise sample, but the learned branch used at deployment is the average-flow predictor $u_\theta$. This is the mechanism that allows infer1, infer2, and infer3 policies to compete with infer100 diffusion-style baselines.
\subsection{Attention Residual policy transformer}
\label{sec:attnres_policy}
We spell out AttnRes by first rewriting an ordinary residual stack. Let $x_l$ be the hidden state entering layer $l$, and let the layer output increment be
\begin{equation}
y_l = f_l(x_{l-1}), \qquad x_l = x_{l-1}+y_l .
\end{equation}
With the convention $y_0=x_0$, recursively expanding the residual updates gives
\begin{equation}
x_l = y_0+y_1+\cdots+y_l = \sum_{s=0}^{l} y_s .
\end{equation}
Therefore, the next residual block can be written as
\begin{equation}
y_{l+1}=f_{l+1}(x_l)
= f_{l+1}\left(\sum_{s=0}^{l} y_s\right).
\label{eq:standard_residual_sum}
\end{equation}
This makes the implicit assumption clear: a standard residual network gives every previous increment a fixed coefficient of one before layer $l+1$ computes the next increment. If we normalize the coefficients only to emphasize their relative importance, this is equivalent to uniform depth aggregation,
\begin{equation}
y_{l+1}=f_{l+1}\left((l+1)\sum_{s=0}^{l}\frac{1}{l+1}y_s\right),
\end{equation}
so the layer cannot choose which earlier residual features are most useful.
AttnRes replaces this fixed aggregation with learned, layer-dependent weights. Instead of feeding $f_{l+1}$ the unweighted residual sum, we form
\begin{equation}
\tilde{x}_l = \sum_{s=0}^{l} a_{l+1,s} y_s,
\qquad
a_{l+1,s}\ge 0,
\qquad
\sum_{s=0}^{l} a_{l+1,s}=1,
\label{eq:weighted_residual_sum}
\end{equation}
then compute
\begin{equation}
y_{l+1}=f_{l+1}(\tilde{x}_l),
\qquad
x_{l+1}=x_l+y_{l+1}.
\end{equation}
The attention weights are obtained by scoring each previous residual increment with a learned query vector for the current layer:
\begin{equation}
e_{l+1,s}=w_{l+1}^{\top}\mathrm{RMSNorm}(y_s),
\qquad
a_{l+1,s}=\frac{\exp(e_{l+1,s})}{\sum_{j=0}^{l}\exp(e_{l+1,j})}.
\label{eq:attnres_weights}
\end{equation}
Equations~\eqref{eq:standard_residual_sum}--\eqref{eq:attnres_weights} show the difference in one line: standard residuals use a fixed sum $\sum_s y_s$, whereas AttnRes uses an attention-weighted sum $\sum_s a_{l+1,s}y_s$ whose coefficients depend on the current layer. This keeps the residual pathway but lets each policy layer select earlier visual-state-action features rather than uniformly accumulating all previous increments. In our object-transfer ablations, the best settings apply AttnRes in the policy transformer; replacing residuals throughout the vision encoder underperforms, suggesting that the placement of learned residual aggregation is important.
\subsection{Inference and control loop}
\label{sec:inference_loop}
At deployment, the iMF-AttnRes policy samples an action chunk with one to a few average-flow evaluations and then executes the chunk for \texttt{exec} control steps. This contrasts with diffusion-style baselines whose run names include infer100. For socket insertion, the strongest iMF-AttnRes policies use ph32/exec16 and infer2 or infer3. For object transfer, the strongest iMF-AttnRes run uses ph32/exec16 and infer1. The resulting control loop is simple: encode observations, run the AttnRes transformer, evaluate the iMF action generator a small number of times, execute the predicted chunk, and replan on the next observation window.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-method-overview}}]
\small Paper Banana prompt: create a clean 16:9 academic system diagram for iMF-AttnRes VLA. Show multi-view images, robot state, and optional language/task conditioning entering a VLA encoder; a policy transformer with AttnRes depth-wise residual aggregation; an improved Mean Flow action generator predicting average flow with one-to-few evaluations; and an exec-length action chunk controlling RoboIMI socket-insert and sim\_transfer robots. Use muted conference-paper colors, clear arrows, no decorative 3D, and labels for iMF, AttnRes, infer1/infer2/infer3, and exec16.
\end{tcolorbox}
\caption{Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image.}
\label{fig:method_overview}
\end{figure}
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-imf-attnres-components}}]
\small Paper Banana prompt: create a 16:9 technical diagram contrasting classical flow matching instantaneous velocity $dx_t/dt=v(x_t,t)$; Mean Flow average velocity over $[r,t]$ with the JVP correction term; improved Mean Flow training relation $V_\theta(z_t)=u_\theta(z_t)+(t-r)\mathrm{JVP}_{\mathrm{sg}}(u_\theta;v_\theta)$; and AttnRes weighted residual aggregation $a_{t+1,s}$ proportional to $\exp(w_{t+1}\cdot\mathrm{RMSNorm}(y_s))$. Use equation callouts, minimal arrows, and a final callout saying one-to-few-step action generation for robot control.
\end{tcolorbox}
\caption{Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering.}
\label{fig:imf_attnres_components}
\end{figure}
\section{Experiments}
\label{sec:experiments}
\subsection{Setup and metrics}
\label{sec:setup_metrics}
We evaluate in two RoboIMI simulation environments. The socket peg task measures progress toward inserting a peg-like object into a socket. The sim\_transfer task measures object-transfer manipulation. Each reported row uses 100 rollouts. Metrics are copied directly from the rollout logs: average cumulative reward, median reward, mean maximum reward, maximum cumulative reward, count of nonzero-reward episodes, count of success-like episodes, inference FPS, control FPS, and average inference time when available. For socket insertion, the recorded success-like threshold is \texttt{max\_reward > 4}; for sim\_transfer, it is \texttt{max\_reward >= 4}. Hardware differs across runs, so speed comparisons are practical deployment measurements rather than perfectly normalized throughput benchmarks.
\subsection{Socket peg insertion}
\label{sec:socket_results}
\Cref{tab:socket_results} summarizes the socket peg insertion results. The best iMF-AttnRes policy by average reward is the infer3 model, with average reward 1513.56 and median reward 1901.5. It exceeds both diffusion-style infer100 baselines: the ph16 baseline obtains average reward 861.53 and median reward 668.0, while the ph32/exec16 baseline obtains average reward 1019.39 and median reward 804.0. The latency gap is large. The iMF-AttnRes infer2 and infer3 policies require 14.409 ms and 15.122 ms per inference, while the two infer100 baselines require 338.291 ms and 397.912 ms.
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{3.5pt}
\caption{Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt{max\_reward > 4} count.}
\label{tab:socket_results}
\begin{tabular}{llrrrrrr}
\toprule
Method & Setting & Avg. reward & Median & Avg. max & Nonzero & Success-like & Latency ms \\
\midrule
iMF-AttnRes & infer1, ph32, exec16, 50k & 1275.22 & 1490.5 & 3.12 & 84/100 & 0/100 & 16.186 \\
Diffusion-style VLA & infer100, ph16, 150k & 861.53 & 668.0 & 2.58 & 97/100 & 2/100 & 338.291 \\
Diffusion-style VLA & infer100, ph32, exec16, 150k & 1019.39 & 804.0 & 2.47 & 92/100 & 4/100 & 397.912 \\
iMF-AttnRes & infer2, ph32, exec16, 150k & 1472.62 & 1825.0 & 3.29 & 83/100 & 5/100 & 14.409 \\
iMF-AttnRes & infer3, ph32, exec16, 150k & \textbf{1513.56} & \textbf{1901.5} & 3.28 & 83/100 & 3/100 & 15.122 \\
ACT & action chunking & 289.60 & 13.0 & 1.29 & 55/100 & 1/100 & 100.212 \\
SmolVLA & 100k & 466.16 & 106.0 & 1.72 & 89/100 & 0/100 & \textbf{2.614} \\
\bottomrule
\end{tabular}
\end{table*}
The nonzero-reward count reveals an important caveat. The ph16 diffusion-style baseline has 97/100 nonzero episodes, higher than the iMF-AttnRes infer2 and infer3 policies at 83/100. Yet its average and median rewards are much lower. Thus, in this task, nonzero contact or partial progress is not sufficient; the iMF-AttnRes policies more often produce high-reward trajectories when they engage the task successfully. SmolVLA is the fastest method in raw latency and control FPS, but its reward is lower than iMF-AttnRes. ACT underperforms both iMF-AttnRes and the diffusion-style VLA baselines in this setup.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-socket-reward-latency}}]
\small Paper Banana prompt: create a 4:3 publication-quality reward-latency comparison for socket peg insertion. Plot methods from the socket table with average reward on the y-axis and average inference time or inverse latency on the x-axis. Highlight iMF-AttnRes infer2/infer3 at 1472.62/1513.56 average reward and 14.409/15.122 ms, diffusion-style infer100 baselines at 861.53/1019.39 average reward and 338.291/397.912 ms, ACT at 289.6 and 100.2116 ms, and SmolVLA at 466.16 and 2.614 ms. Use log-scale latency if helpful and label the speed-quality Pareto frontier.
\end{tcolorbox}
\caption{Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image.}
\label{fig:socket_reward_latency}
\end{figure}
\subsection{Object transfer and ablations}
\label{sec:sim_transfer_results}
\Cref{tab:sim_transfer_results} reports the object-transfer results. The strongest iMF-AttnRes run reaches average reward 526.22 and 44/100 success-like episodes, exceeding the native diffusion-policy DiT/DDPM/ResNet baseline at average reward 319.2 and 29/100 success-like episodes. This is the clearest sim\_transfer evidence that iMF-AttnRes can improve both reward and success-like threshold count.
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{3.2pt}
\caption{Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt{max\_reward >= 4} count.}
\label{tab:sim_transfer_results}
\begin{tabular}{llrrrrrr}
\toprule
Method & Setting & Avg. reward & Median & Avg. max & Nonzero & Success-like & Inf. FPS \\
\midrule
Native Diffusion Policy & DiT + DDPM + ResNet & 319.20 & 6.0 & 1.68 & 55/100 & 29/100 & 32.09 \\
Diffusion-style VLA & emb384, layer18 & 233.52 & 0.0 & 1.12 & 39/100 & 17/100 & 1.859 \\
iMF multi-token ResNet18 & step34999, ph16, exec08 & 260.66 & 0.0 & 1.12 & 33/100 & 23/100 & \textbf{441.80} \\
iMF full AttnRes vision & ph16, exec08, 50k & 228.42 & 0.0 & 0.94 & 31/100 & 16/100 & 55.997 \\
iMF-AttnRes DiT only & ph16, exec16, 50k & 240.64 & 0.0 & 1.02 & 32/100 & 19/100 & 137.996 \\
iMF-AttnRes DiT only & ph32, exec08, 50k & 163.28 & 0.0 & 0.76 & 26/100 & 12/100 & 85.638 \\
iMF-AttnRes DiT only & ph32, exec32, 50k & 260.72 & 0.0 & 1.18 & 38/100 & 21/100 & 9.557 \\
iMF-AttnRes DiT only & ph16, exec08, 50k & 229.56 & 0.0 & 1.18 & 41/100 & 18/100 & 69.984 \\
iMF-AttnRes DiT only & ph08, exec08, 50k & 237.02 & 0.0 & 1.32 & 48/100 & 18/100 & 69.192 \\
iMF-AttnRes DiT only & ph32, exec16, 50k & 49.88 & 0.0 & 0.32 & 13/100 & 4/100 & 138.903 \\
iMF-AttnRes & infer1, ph32, exec16, 50k & \textbf{526.22} & \textbf{86.0} & \textbf{2.14} & \textbf{63/100} & \textbf{44/100} & 8.911 \\
\bottomrule
\end{tabular}
\end{table*}
The ablations are mixed and therefore useful. The ResNet18 multi-token iMF model is extremely fast at 441.80 inference FPS, but its average reward is 260.66, below the native diffusion-policy baseline. The full-AttnRes vision model reaches 228.42 average reward, suggesting that replacing residuals inside the vision encoder is not automatically helpful. Horizon and execution length are also sensitive: the ph32/exec16 DiT-only iMF variant reaches only 49.88 average reward despite high inference FPS. The strongest object-transfer run therefore combines iMF-AttnRes with the right horizon and execution setting rather than showing a universally dominant architectural change.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-sim-transfer-ablation}}]
\small Paper Banana prompt: create a 4:3 grouped bar chart for sim\_transfer ablations. Show average reward and success-like episode count for native diffusion policy, best sim-transfer iMF-AttnRes infer1, ResNet18 multi-token iMF, full-AttnRes vision, and selected horizon/execution variants. Emphasize that best iMF-AttnRes reaches 526.22 average reward and 44/100 success-like episodes versus native diffusion policy at 319.2 and 29/100, while several ablations underperform.
\end{tcolorbox}
\caption{Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering.}
\label{fig:sim_transfer_ablation}
\end{figure}
\subsection{Derived speed-quality comparisons}
\label{sec:derived_comparisons}
\Cref{tab:derived_comparisons} lists the derived comparisons used to summarize the tradeoff. On socket insertion, iMF-AttnRes infer3 improves average reward by 1.76$\times$ over the ph16 diffusion-style baseline and by 1.48$\times$ over the ph32 diffusion-style baseline. The same infer3 run improves inference FPS by 22.97$\times$ over the ph16 diffusion-style baseline. The infer2 run is 23.48$\times$ lower latency than the ph16 diffusion-style baseline and 28.45$\times$ higher inference FPS than the ph32 diffusion-style baseline. On object transfer, best iMF-AttnRes improves average reward by 1.65$\times$ and success-like episodes by +15 over native Diffusion Policy.
\begin{table}[t]
\centering
\small
\caption{Derived comparisons from the 100-rollout logs.}
\label{tab:derived_comparisons}
\begin{tabular}{llr}
\toprule
Comparison & Metric & Result \\
\midrule
Socket iMF infer3 vs ph16 diffusion & Avg. reward & 1.76$\times$ \\
Socket iMF infer3 vs ph32 diffusion & Avg. reward & 1.48$\times$ \\
Socket iMF infer3 vs ACT & Avg. reward & 5.23$\times$ \\
Socket iMF infer3 vs SmolVLA & Avg. reward & 3.25$\times$ \\
Socket iMF infer2 vs ph16 diffusion & Latency reduction & 23.48$\times$ \\
Socket iMF infer3 vs ph16 diffusion & Inference FPS & 22.97$\times$ \\
Socket iMF infer2 vs ph32 diffusion & Inference FPS & 28.45$\times$ \\
Sim transfer iMF vs native diffusion & Avg. reward & 1.65$\times$ \\
Sim transfer iMF vs native diffusion & Success-like episodes & +15 \\
ResNet18 iMF vs layer18 baseline & Inference FPS & 237.65$\times$ \\
\bottomrule
\end{tabular}
\end{table}
\section{Limitations}
\label{sec:limitations}
All experiments are simulation rollouts. The data do not establish real-robot transfer, robustness to sensing changes, or safety under hardware execution. CoRL submissions should provide convincing robotics evidence; the present draft should therefore be viewed as a simulation-first manuscript that still needs real-robot or stronger transfer evidence before formal submission.
The speed measurements are also not perfectly hardware-normalized. Runs were collected on RTX 5880 Ada, L20, and RTX 5090 machines as encoded by their run names. Large latency differences between infer100 diffusion-style policies and infer1--3 iMF-AttnRes policies are meaningful because they reflect algorithmic sampling cost, but exact FPS ratios may include hardware and implementation effects. Future experiments should rerun all main policies on the same GPU, with identical rollout parallelism and identical observation preprocessing.
Finally, iMF-AttnRes is sensitive to design choices. In sim\_transfer, several iMF variants underperform the native diffusion-policy baseline, and full AttnRes in the vision encoder performs worse than more conservative policy-transformer AttnRes. This suggests that average-flow training is not a plug-in guarantee; horizon, execution length, visual tokenization, and residual placement must be tuned carefully.
\section{Conclusion}
\label{sec:conclusion}
We presented a first simulation study of iMF-AttnRes for fast robot action generation. The method adapts improved Mean Flow to VLA imitation learning and uses Attention Residuals to replace fixed residual accumulation in the policy transformer. In socket peg insertion, iMF-AttnRes achieves higher average and median reward than diffusion-style infer100 baselines while reducing inference latency from hundreds of milliseconds to roughly 15 ms. In object transfer, the best iMF-AttnRes run improves average reward and success-like episode count over the native diffusion-policy baseline. At the same time, ablations show that the method is sensitive to horizon/execution settings and residual placement. The next step is controlled, hardware-normalized evaluation with real-robot transfer.
\clearpage
\bibliography{refs}
\end{document}
+140
View File
@@ -0,0 +1,140 @@
@article{chi2023diffusion,
title = {Diffusion Policy: Visuomotor Policy Learning via Action Diffusion},
author = {Chi, Cheng and Xu, Zhenjia and Feng, Siyuan and Cousineau, Eric and Du, Yilun and Burchfiel, Benjamin and Tedrake, Russ and Song, Shuran},
year = {2023},
journal = {arXiv preprint arXiv:2303.04137},
eprint = {2303.04137},
archivePrefix = {arXiv}
}
@inproceedings{ho2020denoising,
title = {Denoising Diffusion Probabilistic Models},
author = {Ho, Jonathan and Jain, Ajay and Abbeel, Pieter},
year = {2020},
booktitle = {Advances in Neural Information Processing Systems}
}
@inproceedings{peebles2023scalable,
title = {Scalable Diffusion Models with Transformers},
author = {Peebles, William and Xie, Saining},
year = {2023},
booktitle = {IEEE/CVF International Conference on Computer Vision}
}
@inproceedings{lipman2023flow,
title = {Flow Matching for Generative Modeling},
author = {Lipman, Yaron and Chen, Ricky T. Q. and Ben-Hamu, Heli and Nickel, Maximilian and Le, Matt},
year = {2023},
booktitle = {International Conference on Learning Representations}
}
@article{liu2022flow,
title = {Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow},
author = {Liu, Xingchao and Gong, Chengyue and Liu, Qiang},
year = {2022},
journal = {arXiv preprint arXiv:2209.03003},
eprint = {2209.03003},
archivePrefix = {arXiv}
}
@article{geng2025mean,
title = {Mean Flows for One-step Generative Modeling},
author = {Geng, Zhengyang and Deng, Mingyang and Bai, Xingjian and Kolter, J. Zico and He, Kaiming},
year = {2025},
journal = {arXiv preprint arXiv:2505.13447},
eprint = {2505.13447},
archivePrefix = {arXiv}
}
@article{geng2025improved,
title = {Improved Mean Flows: On the Challenges of Fastforward Generative Models},
author = {Geng, Zhengyang and Lu, Yiyang and Wu, Zongze and Shechtman, Eli and Kolter, J. Zico and He, Kaiming},
year = {2025},
journal = {arXiv preprint arXiv:2512.02012},
eprint = {2512.02012},
archivePrefix = {arXiv}
}
@article{brohan2022rt,
title = {RT-1: Robotics Transformer for Real-World Control at Scale},
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Dabis, Joseph and Finn, Chelsea and Gopalakrishnan, Keerthana and Hausman, Karol and Herzog, Alexander and Hsu, Jasmine and others},
year = {2022},
journal = {arXiv preprint arXiv:2212.06817},
eprint = {2212.06817},
archivePrefix = {arXiv}
}
@inproceedings{brohan2023rt,
title = {RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control},
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Chen, Xi and Choromanski, Krzysztof and Ding, Tianli and Driess, Danny and Dubey, Avinava and Finn, Chelsea and others},
year = {2023},
booktitle = {Conference on Robot Learning}
}
@article{kim2024openvla,
title = {OpenVLA: An Open-Source Vision-Language-Action Model},
author = {Kim, Moo Jin and Pertsch, Karl and Karamcheti, Siddharth and Xiao, Ted and Balakrishna, Ashwin and Nair, Suraj and Rafailov, Rafael and Foster, Ethan and Lam, Grace and Sanketi, Pannag and others},
year = {2024},
journal = {arXiv preprint arXiv:2406.09246},
eprint = {2406.09246},
archivePrefix = {arXiv}
}
@article{shukor2025smolvla,
title = {SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},
author = {Shukor, Mustafa and Aubakirova, Dana and Capuano, Francesco and Kooijmans, Pepijn and Palma, Steven and Zouitine, Adil and Aractingi, Michel and Pascal, Caroline and Russi, Martino and Marafioti, Andres and others},
year = {2025},
journal = {arXiv preprint arXiv:2506.01844},
eprint = {2506.01844},
archivePrefix = {arXiv}
}
@article{black2024pi0,
title = {{$\pi_0$}: A Vision-Language-Action Flow Model for General Robot Control},
author = {Black, Kevin and Brown, Noah and Driess, Danny and Esmail, Adnan and Equi, Michael and Finn, Chelsea and Fusai, Niccolo and Groom, Lachy and Hausman, Karol and Ichter, Brian and others},
year = {2024},
journal = {arXiv preprint arXiv:2410.24164},
eprint = {2410.24164},
archivePrefix = {arXiv}
}
@article{pertsch2025fast,
title = {FAST: Efficient Action Tokenization for Vision-Language-Action Models},
author = {Pertsch, Karl and Stachowicz, Kyle and Ichter, Brian and Driess, Danny and Nair, Suraj and Vuong, Quan and Mees, Oier and Finn, Chelsea and Levine, Sergey},
year = {2025},
journal = {arXiv preprint arXiv:2501.09747},
eprint = {2501.09747},
archivePrefix = {arXiv}
}
@article{zhao2023learning,
title = {Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware},
author = {Zhao, Tony Z. and Kumar, Vikash and Levine, Sergey and Finn, Chelsea},
year = {2023},
journal = {arXiv preprint arXiv:2304.13705},
eprint = {2304.13705},
archivePrefix = {arXiv}
}
@inproceedings{he2016deep,
title = {Deep Residual Learning for Image Recognition},
author = {He, Kaiming and Zhang, Xiangyu and Ren, Shaoqing and Sun, Jian},
year = {2016},
booktitle = {IEEE Conference on Computer Vision and Pattern Recognition}
}
@inproceedings{vaswani2017attention,
title = {Attention Is All You Need},
author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N. and Kaiser, Lukasz and Polosukhin, Illia},
year = {2017},
booktitle = {Advances in Neural Information Processing Systems}
}
@article{kimi2026attention,
title = {Attention Residuals},
author = {{Kimi Team}},
year = {2026},
journal = {arXiv preprint arXiv:2603.15031},
eprint = {2603.15031},
archivePrefix = {arXiv}
}
+54
View File
@@ -0,0 +1,54 @@
{
"candidates": [
{
"title": "Diffusion Policy",
"snippet": "# Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details will be publicly available.\n[...]\n[width=0.95]figure/DP_teaser.pdf \\ca",
"source_url": "https://arxiv.org/html/2502.12371v1",
"discovered_for": [
"related_work.diffusion_policy"
],
"_exa_id": "https://arxiv.org/html/2502.12371v1",
"_exa_published_date": "2025-01-01T00:00:00.000Z"
},
{
"title": "Visuomotor policy learning via action diffusion - ACM Digital Library",
"snippet": "Diffusion policy: : Visuomotor policy learning via action diffusion: International Journal of Robotics Research: Vol 44, No 10-11 other-periodical;requestedJournal:\n[...]\n:rbrs;wgroup:string:ACM Publication Websites;subPage:string:Basic Abstract;article:article:doi\\:10.\n[...]\n783649241273668;page:string:Article/Chapter View;ctype:string:Journal Content;groupTopic:topic:acm-pubtype>other-periodical;website:website:dl-site;issue:issue:doi\\:10.5555/rbrs.2025.44.issue-10-11;csubtype:string:Periodical;taxonomy:taxonomy:acm-pubtype;pageGroup:string:Publication Pages\"> skip to main content\n\n \n\n \n\nContents\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the p",
"source_url": "https://dl.acm.org/doi/10.1177/02783649241273668",
"discovered_for": [
"related_work.diffusion_policy"
],
"_exa_id": "https://dl.acm.org/doi/10.1177/02783649241273668",
"_exa_published_date": null
},
{
"title": "",
"snippet": "This paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor\n[...]\npolicy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4\n[...]\ndifferent robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning\n[...]\nmethods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score\n[...]\nfunction and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin\n[...]\ndynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including\n[...]\ngracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting\n[...]\nimpressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical\n[...]\nrobots, this paper presents a set of key technical contributions including the incorporation of receding horizon control,\n[...]\nvisual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of\n[...]\npolicy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models.\n[...]\nintroducing a new form of robot visuomotor policy that\n[...]\ngenerates behavior via a “conditional denoising di",
"source_url": "https://arxiv.org/pdf/2303.04137v5",
"discovered_for": [
"related_work.diffusion_policy"
],
"_exa_id": "https://arxiv.org/pdf/2303.04137v5",
"_exa_published_date": null
},
{
"title": "Visuomotor Policy Learning via Action Diffusion - arXiv",
"snippet": "This paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details is available diffusion-policy.cs.columbia.edu\n[...]\nIn this work, we seek to address this challenge by introducing a new form of robot visuomoto",
"source_url": "https://arxiv.org/abs/2303.04137",
"discovered_for": [
"related_work.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2303.04137",
"_exa_published_date": "2023-03-07T00:00:00.000Z"
},
{
"title": "3D Diffusion Policy: Generalizable Visuomotor Policy Learning via Simple 3D Representations",
"snippet": "Imitation learning provides an efficient way to teach robots dexterous skills; however, learning complex skills robustly and generalizablely usually consumes large amounts of human demonstrations. To tackle this challenging problem, we present 3D Diffusion Policy (DP3), a novel visual imitation learning approach that incorporates the power of 3D visual representations into diffusion policies, a class of conditional action generative models. The core design of DP3 is the utilization of a compact 3D visual representation, extracted from sparse point clouds with an efficient point encoder. In our experiments involving 72 simulation tasks, DP3 successfully handles most tasks with just 10 demonstrations and surpasses baselines with a 24.2% relative improvement. In 4 real robot tasks, DP3 demonstrates precise control with a high success rate of 85%, given only 40 demonstrations of each task, and shows excellent generalization abilities in diverse aspects, including space, viewpoint, appearance, and instance. Interestingly, in real robot experiments, DP3 rarely violates safety requirements, in contrast to baseline methods which frequently do, necessitating human intervention. Our extensive evaluation highlights the critical importance of 3D representations in real-world robot learning. Videos, code, and data are available on 3d-diffusion-policy.github.io.\n[...]\nTo tackle this challenging problem, we introduce 3D Diffusion Policy (DP3), a simple yet effective visual imitation learnin",
"source_url": "https://arxiv.org/html/2403.03954v2",
"discovered_for": [
"related_work.diffusion_policy"
],
"_exa_id": "https://arxiv.org/html/2403.03954v2",
"_exa_published_date": null
}
]
}
+6
View File
@@ -0,0 +1,6 @@
{
"fig_method_overview": "Paper Banana prompt placeholder: create a clean 16:9 academic system diagram for iMF-AttnRes VLA. Show multi-view images, robot state, and optional language/task conditioning entering a VLA encoder; a policy transformer with AttnRes depth-wise residual aggregation; an improved Mean Flow action generator predicting average flow with one-to-few evaluations; and an exec-length action chunk controlling RoboIMI socket-insert and sim_transfer robots. Use muted conference-paper colors, clear arrows, no decorative 3D, and labels for iMF, AttnRes, infer1/infer2/infer3, and exec16.",
"fig_socket_reward_latency": "Paper Banana prompt placeholder: create a 4:3 publication-quality reward-latency comparison for socket peg insertion. Plot methods from experimental_log Table 1 with avg_reward on the y-axis and avg_inference_time_ms or inverse latency on the x-axis. Highlight iMF-AttnRes infer2/infer3 at 1472.62/1513.56 avg_reward and 14.409/15.122 ms, diffusion-style infer100 baselines at 861.53/1019.39 avg_reward and 338.291/397.912 ms, ACT at 289.6 and 100.2116 ms, and SmolVLA at 466.16 and 2.614 ms. Use log-scale latency if helpful and label the speed-quality Pareto frontier.",
"fig_sim_transfer_ablation": "Paper Banana prompt placeholder: create a 4:3 grouped bar chart for sim_transfer ablations from experimental_log Table 2. Show avg_reward and success_like episode count for native diffusion policy, best sim-transfer iMF-AttnRes infer1, ResNet18 multi-token iMF, full-AttnRes vision, and selected horizon/execution variants. Emphasize that best iMF-AttnRes reaches 526.22 avg_reward and 44/100 success-like episodes versus native diffusion policy at 319.2 and 29/100, while several ablations underperform.",
"fig_imf_attnres_components": "Paper Banana prompt placeholder: create a 16:9 technical diagram contrasting four mathematical components: classical flow matching instantaneous velocity dx_t/dt=v(x_t,t); Mean Flow average velocity over [r,t] with the JVP correction term; improved Mean Flow training relation V_theta(z_t)=u_theta(z_t)+(t-r)JVP_sg(u_theta;v_theta); and AttnRes weighted residual aggregation a_{t+1,s} proportional to exp(w_{t+1} dot RMSNorm(y_s)). Use equation callouts, minimal arrows, and a final callout saying one-to-few-step action generation for robot control."
}
@@ -0,0 +1 @@
Paper Banana prompt placeholder: create a 16:9 technical diagram contrasting four mathematical components: classical flow matching instantaneous velocity dx_t/dt=v(x_t,t); Mean Flow average velocity over [r,t] with the JVP correction term; improved Mean Flow training relation V_theta(z_t)=u_theta(z_t)+(t-r)JVP_sg(u_theta;v_theta); and AttnRes weighted residual aggregation a_{t+1,s} proportional to exp(w_{t+1} dot RMSNorm(y_s)). Use equation callouts, minimal arrows, and a final callout saying one-to-few-step action generation for robot control.
@@ -0,0 +1 @@
Paper Banana prompt placeholder: create a clean 16:9 academic system diagram for iMF-AttnRes VLA. Show multi-view images, robot state, and optional language/task conditioning entering a VLA encoder; a policy transformer with AttnRes depth-wise residual aggregation; an improved Mean Flow action generator predicting average flow with one-to-few evaluations; and an exec-length action chunk controlling RoboIMI socket-insert and sim_transfer robots. Use muted conference-paper colors, clear arrows, no decorative 3D, and labels for iMF, AttnRes, infer1/infer2/infer3, and exec16.
@@ -0,0 +1 @@
Paper Banana prompt placeholder: create a 4:3 grouped bar chart for sim_transfer ablations from experimental_log Table 2. Show avg_reward and success_like episode count for native diffusion policy, best sim-transfer iMF-AttnRes infer1, ResNet18 multi-token iMF, full-AttnRes vision, and selected horizon/execution variants. Emphasize that best iMF-AttnRes reaches 526.22 avg_reward and 44/100 success-like episodes versus native diffusion policy at 319.2 and 29/100, while several ablations underperform.
@@ -0,0 +1 @@
Paper Banana prompt placeholder: create a 4:3 publication-quality reward-latency comparison for socket peg insertion. Plot methods from experimental_log Table 1 with avg_reward on the y-axis and avg_inference_time_ms or inverse latency on the x-axis. Highlight iMF-AttnRes infer2/infer3 at 1472.62/1513.56 avg_reward and 14.409/15.122 ms, diffusion-style infer100 baselines at 861.53/1019.39 avg_reward and 338.291/397.912 ms, ACT at 289.6 and 100.2116 ms, and SmolVLA at 466.16 and 2.614 ms. Use log-scale latency if helpful and label the speed-quality Pareto frontier.
Binary file not shown.
+472
View File
@@ -0,0 +1,472 @@
% File: corl_2026.sty
%
% Latex templates for the Conference on Robot Learning (CoRL)
%
% This template is heavily inspired by the NeurIPS, ICML, ICLR and IEEE Transactions latex templates.
% Hence we would like to thank: Roman Garnett and the previous mantainers of the NIPS style, Percy Liang and the previous mantainers of the ICML style, Hugo Larochelle for the ICLR style, and Michael Shell for the IEEE Transactions style.
%
% History:
% 2017/04/16 - First revision by Roberto Calandra (roberto.calandra@berkeley.edu).
% Main changes:
% - The abstract is more compact compared to NeurIPS/ICML
% - References are by default using natbib with squared numbers (e.g., [1])
% - DOI fields from the bibtex are automatically converted to hyperlinks
% to the corresponding page
% - acknowledgments are now a command, and the corresponding subsubsection is
% automatically included only in the final version
% 2017/06/12 - Modified to use corlabbrvnat.bst, which order the reference by order of appearance in the paper
% 2017/06/13 - fixed typo
% 2018/05/09 - Slightly modified for CoRL 2018 by Jun Morimoto (xmorimo@atr.jp)
% 2019/01/28 - Slightly modified for CoRL 2019 by Jun Nakanishi (jnakanis@meijo-u.ac.jp)
% 2020/02/02 - Slightly modified for CoRL 2020 by Cynthia Matuszek (cmat@umbc.edu)
% 2020/08/19 - Added preprint option by Roberto Calandra (rcalandra@fb.com)
% 2021/05/06 - Slightly modified for CoRL 2021 by Gerhard Neumann (gerhard.neumann@kit.edu)
% 2022/03/09 - Slightly modified for CoRL 2022 by Minas Liarokapis (minas.liarokapis@auckland.ac.nz)
% 2022/03/06 - Slightly modified for CoRL 2023 by Marc Toussaint (toussaint@tu-berlin.de)
% 2024/03/26 - Slightly modified for CoRL 2024 by David Held (dheld@andrew.cmu.edu)
% 2026/01/10 - Slightly modified for CoRL 2026 by Yoonchang Sung (yoonchang.sung@ntu.edu.sg)
%
% TODO: nohyperref is not working at the moment
%
\NeedsTeXFormat{LaTeX2e}
% Content to be changed from year to year
\ProvidesPackage{corl_2026}[2026/08/15 CORL2026 submission/preprint/camera-ready style file]
\newcommand{\@conferenceordinal}{10th}
\newcommand{\@conferenceyear}{2026}
\newcommand{\@conferencelocation}{Austin TX, USA}
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Accepted options: [final,preprint,nonatbib,nohyperref]
% Declare the final option, which creates camera-ready copy
\newif\if@conferencefinal\@conferencefinalfalse
\DeclareOption{final}{
\@conferencefinaltrue
}
% Declare the preprint option, which creates a camera-ready copy without the corl footnote
\newif\if@preprinttype\@preprinttypefalse
\DeclareOption{preprint}{
\@preprinttypetrue
}
% The natbib package is loaded by default. Declaring the nonatbib option, does not load natbib in case of package clash (users can pass options to natbib via \PassOptionsToPackage)
\newif\if@natbib\@natbibtrue
\DeclareOption{nonatbib}{
\@natbibfalse
}
% The hyperref package is loaded by default. Declaring the nohyperref option, does not load the hyperref.
\DeclareOption{nohyperref}{%
\gdef\nohyperref{1}
}
% Activate the options
\ProcessOptions\relax
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Required packages:
\RequirePackage{lineno}
\RequirePackage{color}
% Load natbib unless told otherwise
\if@natbib
\RequirePackage[square,numbers]{natbib}
\bibliographystyle{corlabbrvnat}
\fi
% set page geometry
\RequirePackage{hyperref} % hyperlinks
\RequirePackage[verbose=true,letterpaper]{geometry}
\AtBeginDocument{
\newgeometry{
textheight=9in,
textwidth=5.5in,
top=1in,
headheight=12pt,
headsep=25pt,
footskip=30pt
}
\@ifpackageloaded{fullpage}
{\PackageWarning{corl_2026}{fullpage package not allowed! Overwriting formatting.}}
{}
}
\ifdefined\nohyperref\else\ifdefined\hypersetup
\definecolor{mydarkblue}{rgb}{0,0.08,0.45}
\hypersetup{ %
pdftitle={},
pdfauthor={},
pdfsubject={Proceedings of the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear)},
pdfkeywords={},
pdfborder=0 0 0,
pdfpagemode=UseNone,
colorlinks=true,
linkcolor=mydarkblue,
citecolor=mydarkblue,
filecolor=mydarkblue,
urlcolor=mydarkblue,
pdfview=FitH}
\ifdefined\isaccepted \else
\hypersetup{pdfauthor={Anonymous Submission}}
\fi
\fi\fi
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%
% fonts
\renewcommand{\rmdefault}{ptm}
\renewcommand{\sfdefault}{phv}
% Create acknowledgments -- only if the option 'final' is activated
\providecommand{\acknowledgments}{}
\renewcommand{\acknowledgments}[1]{%
\if@conferencefinal%
\subsubsection*{Acknowledgments} #1
\fi
\if@preprinttype%
\subsubsection*{Acknowledgments} #1
\fi
}
% handle tweaks for camera-ready copy vs. submission copy
\if@conferencefinal
\newcommand{\@noticestring}{%
\@conferenceordinal\/ Conference on Robot Learning
(CoRL \@conferenceyear), \@conferencelocation.%
}
\else
\if@preprinttype
\newcommand{\@noticestring}{%
% Nothing here.
}
\else
\newcommand{\@noticestring}{%
Submitted to the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear). Do not distribute.%
}
% line numbers for submission
\linenumbers
% fix incompatibilities between lineno and amsmath, if required, by
% transparently wrapping linenomath environments around amsmath
% environments
\AtBeginDocument{%
\@ifpackageloaded{amsmath}{%
\newcommand*\patchAmsMathEnvironmentForLineno[1]{%
\expandafter\let\csname old#1\expandafter\endcsname\csname #1\endcsname
\expandafter\let\csname oldend#1\expandafter\endcsname\csname end#1\endcsname
\renewenvironment{#1}%
{\linenomath\csname old#1\endcsname}%
{\csname oldend#1\endcsname\endlinenomath}%
}%
\newcommand*\patchBothAmsMathEnvironmentsForLineno[1]{%
\patchAmsMathEnvironmentForLineno{#1}%
\patchAmsMathEnvironmentForLineno{#1*}%
}%
\patchBothAmsMathEnvironmentsForLineno{equation}%
\patchBothAmsMathEnvironmentsForLineno{align}%
\patchBothAmsMathEnvironmentsForLineno{flalign}%
\patchBothAmsMathEnvironmentsForLineno{alignat}%
\patchBothAmsMathEnvironmentsForLineno{gather}%
\patchBothAmsMathEnvironmentsForLineno{multline}%
}{}
}
\fi
\fi
% The DOI will now automatically generate a URL link. Pretty cool!
% + Fix for hyperref and DOI (https://www.tug.org/pipermail/tex-live/2012-August/032161.html)
%-----------------------------
\makeatletter
\providecommand{\doi}[1]{%
\begingroup
\let\bibinfo\@secondoftwo
\urlstyle{rm}%
\href{http://dx.doi.org/#1}{%
doi:\discretionary{}{}{}%
\nolinkurl{#1}%
}%
\endgroup
}
% \makeatother
% %-----------------------------
\widowpenalty=10000
\clubpenalty=10000
\flushbottom
\sloppy
% font sizes with reduced leading
\renewcommand{\normalsize}{%
\@setfontsize\normalsize\@xpt\@xipt
\abovedisplayskip 7\p@ \@plus 2\p@ \@minus 5\p@
\abovedisplayshortskip \z@ \@plus 3\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 4\p@ \@plus 3\p@ \@minus 3\p@
}
\normalsize
\renewcommand{\small}{%
\@setfontsize\small\@ixpt\@xpt
\abovedisplayskip 6\p@ \@plus 1.5\p@ \@minus 4\p@
\abovedisplayshortskip \z@ \@plus 2\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 3\p@ \@plus 2\p@ \@minus 2\p@
}
\renewcommand{\footnotesize}{\@setfontsize\footnotesize\@ixpt\@xpt}
\renewcommand{\scriptsize}{\@setfontsize\scriptsize\@viipt\@viiipt}
\renewcommand{\tiny}{\@setfontsize\tiny\@vipt\@viipt}
\renewcommand{\large}{\@setfontsize\large\@xiipt{14}}
\renewcommand{\Large}{\@setfontsize\Large\@xivpt{16}}
\renewcommand{\LARGE}{\@setfontsize\LARGE\@xviipt{20}}
\renewcommand{\huge}{\@setfontsize\huge\@xxpt{23}}
\renewcommand{\Huge}{\@setfontsize\Huge\@xxvpt{28}}
% sections with less space
\providecommand{\section}{}
\renewcommand{\section}{%
\@startsection{section}{1}{\z@}%
{-2.0ex \@plus -0.5ex \@minus -0.2ex}%
{ 1.5ex \@plus 0.3ex \@minus 0.2ex}%
{\large\bf\raggedright}%
}
\providecommand{\subsection}{}
\renewcommand{\subsection}{%
\@startsection{subsection}{2}{\z@}%
{-1.8ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.8ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\subsubsection}{}
\renewcommand{\subsubsection}{%
\@startsection{subsubsection}{3}{\z@}%
{-1.5ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.5ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\paragraph}{}
\renewcommand{\paragraph}{%
\@startsection{paragraph}{4}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subparagraph}{}
\renewcommand{\subparagraph}{%
\@startsection{subparagraph}{5}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subsubsubsection}{}
\renewcommand{\subsubsubsection}{%
\vskip5pt{\noindent\normalsize\rm\raggedright}%
}
% float placement
\renewcommand{\topfraction }{0.85}
\renewcommand{\bottomfraction }{0.4}
\renewcommand{\textfraction }{0.1}
\renewcommand{\floatpagefraction}{0.7}
\newlength{\@nipsabovecaptionskip}\setlength{\@nipsabovecaptionskip}{7\p@}
\newlength{\@nipsbelowcaptionskip}\setlength{\@nipsbelowcaptionskip}{\z@}
\setlength{\abovecaptionskip}{\@nipsabovecaptionskip}
\setlength{\belowcaptionskip}{\@nipsbelowcaptionskip}
% swap above/belowcaptionskip lengths for tables
\renewenvironment{table}
{\setlength{\abovecaptionskip}{\@nipsbelowcaptionskip}%
\setlength{\belowcaptionskip}{\@nipsabovecaptionskip}%
\@float{table}}
{\end@float}
% footnote formatting
\setlength{\footnotesep }{6.65\p@}
\setlength{\skip\footins}{9\p@ \@plus 4\p@ \@minus 2\p@}
\renewcommand{\footnoterule}{\kern-3\p@ \hrule width 12pc \kern 2.6\p@}
\setcounter{footnote}{0}
% paragraph formatting
\setlength{\parindent}{\z@}
\setlength{\parskip }{5.5\p@}
% list formatting
\setlength{\topsep }{4\p@ \@plus 1\p@ \@minus 2\p@}
\setlength{\partopsep }{1\p@ \@plus 0.5\p@ \@minus 0.5\p@}
\setlength{\itemsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\parsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\leftmargin }{3pc}
\setlength{\leftmargini }{\leftmargin}
\setlength{\leftmarginii }{2em}
\setlength{\leftmarginiii}{1.5em}
\setlength{\leftmarginiv }{1.0em}
\setlength{\leftmarginv }{0.5em}
\def\@listi {\leftmargin\leftmargini}
\def\@listii {\leftmargin\leftmarginii
\labelwidth\leftmarginii
\advance\labelwidth-\labelsep
\topsep 2\p@ \@plus 1\p@ \@minus 0.5\p@
\parsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\itemsep \parsep}
\def\@listiii{\leftmargin\leftmarginiii
\labelwidth\leftmarginiii
\advance\labelwidth-\labelsep
\topsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\parsep \z@
\partopsep 0.5\p@ \@plus 0\p@ \@minus 0.5\p@
\itemsep \topsep}
\def\@listiv {\leftmargin\leftmarginiv
\labelwidth\leftmarginiv
\advance\labelwidth-\labelsep}
\def\@listv {\leftmargin\leftmarginv
\labelwidth\leftmarginv
\advance\labelwidth-\labelsep}
\def\@listvi {\leftmargin\leftmarginvi
\labelwidth\leftmarginvi
\advance\labelwidth-\labelsep}
% create title
\providecommand{\maketitle}{}
\renewcommand{\maketitle}{%
\par
\begingroup
\renewcommand{\thefootnote}{\fnsymbol{footnote}}
% for perfect author name centering
\renewcommand{\@makefnmark}{\hbox to \z@{$^{\@thefnmark}$\hss}}
% The footnote-mark was overlapping the footnote-text,
% added the following to fix this problem (MK)
\long\def\@makefntext##1{%
\parindent 1em\noindent
\hbox to 1.8em{\hss $\m@th ^{\@thefnmark}$}##1
}
\thispagestyle{empty}
\@maketitle
\@thanks
\@notice
\endgroup
\let\maketitle\relax
\let\thanks\relax
}
% rules for title box at top of first page
\newcommand{\@toptitlebar}{
\hrule height 4\p@
\vskip 0.25in
\vskip -\parskip%
}
\newcommand{\@bottomtitlebar}{
\vskip 0.29in
\vskip -\parskip
\hrule height 1\p@
\vskip 0.09in%
}
%% keywords as first class citizens
\def\keywords#1{%
% \ifdefined\isaccepted \else
% \par {\bf Keywords:} #1%
% \fi
% \ifdefined\nohyperref\else\ifdefined\hypersetup
% \hypersetup{pdfkeywords={#1}}
% \fi\fi
\ifdefined\isaccepted \else
\begin{quote}
\textbf{Keywords:} #1%
\end{quote}
\fi
\ifdefined\nohyperref\else\ifdefined\hypersetup
\hypersetup{pdfkeywords={#1}}
\fi\fi
}
% create title (includes both anonymized and non-anonymized versions)
\providecommand{\@maketitle}{}
\renewcommand{\@maketitle}{%
\vbox{%
\hsize\textwidth
\linewidth\hsize
\vskip 0.1in
% \@toptitlebar
\centering
{\LARGE\bf \@title\par}
% \@bottomtitlebar
\if@conferencefinal
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\if@preprinttype
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}
Anonymous Author(s) \\
Affiliation \\
Address \\
\texttt{email} \\
\end{tabular}%
\fi
\fi
\vskip 0.3in \@minus 0.1in
}
}
% add conference notice to bottom of first page
\newcommand{\ftype@noticebox}{8}
\newcommand{\@notice}{%
% give a bit of extra room back to authors on first page
\enlargethispage{2\baselineskip}%
\@float{noticebox}[b]%
\footnotesize\@noticestring%
\end@float%
}
% abstract styling
\renewenvironment{abstract}%
{%
% \vskip 0.075in%
% \centerline%
% {\large\bf Abstract}%
% \vspace{0.5ex}%
\begin{quote}%
\textbf{Abstract:}%
}
{
\par%
\vskip 1ex%
% \ifdefined\keywords
% \textbf{Keywords:}%
% \@keywords%
% \else
% \fi
\end{quote}%
}
\endinput
File diff suppressed because it is too large Load Diff
+124
View File
@@ -0,0 +1,124 @@
\relax
\bibstyle{corlabbrvnat}
\providecommand\hyper@newdestlabel[2]{}
\providecommand\HyField@AuxAddToFields[1]{}
\providecommand\HyField@AuxAddToCoFields[2]{}
\citation{black2024pi0,chi2023diffusion}
\citation{ho2020denoising}
\citation{peebles2023scalable}
\citation{lipman2023flow,liu2022flow}
\citation{geng2025mean}
\citation{geng2025improved}
\citation{he2016deep,vaswani2017attention}
\citation{kimi2026attention}
\@writefile{toc}{\contentsline {section}{\numberline {1}Introduction}{1}{section.1}\protected@file@percent }
\newlabel{sec:introduction}{{1}{1}{Introduction}{section.1}{}}
\newlabel{sec:introduction@cref}{{[section][1][]1}{[1][1][]1}{}{}{}}
\citation{zhao2023learning,shukor2025smolvla}
\citation{black2024pi0,chi2023diffusion}
\citation{ho2020denoising}
\citation{peebles2023scalable}
\citation{lipman2023flow}
\citation{liu2022flow}
\citation{brohan2022rt,brohan2023rt}
\citation{kim2024openvla}
\citation{shukor2025smolvla,pertsch2025fast}
\citation{zhao2023learning}
\citation{kimi2026attention}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {section}{\numberline {2}Related Work}{2}{section.2}\protected@file@percent }
\newlabel{sec:related_work}{{2}{2}{Related Work}{section.2}{}}
\newlabel{sec:related_work@cref}{{[section][2][]2}{[1][2][]2}{}{}{}}
\@writefile{toc}{\contentsline {paragraph}{Diffusion and flow policies for robot action generation.}{2}{section*.1}\protected@file@percent }
\@writefile{toc}{\contentsline {paragraph}{Vision-language-action models and action chunking.}{2}{section*.2}\protected@file@percent }
\@writefile{toc}{\contentsline {paragraph}{Residual aggregation and attention residuals.}{2}{section*.3}\protected@file@percent }
\citation{geng2025mean}
\citation{geng2025improved}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {section}{\numberline {3}Method}{3}{section.3}\protected@file@percent }
\newlabel{sec:method}{{3}{3}{Method}{section.3}{}}
\newlabel{sec:method@cref}{{[section][3][]3}{[1][2][]3}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.1}Policy formulation}{3}{subsection.3.1}\protected@file@percent }
\newlabel{sec:policy_formulation}{{3.1}{3}{Policy formulation}{subsection.3.1}{}}
\newlabel{sec:policy_formulation@cref}{{[subsection][1][3]3.1}{[1][3][]3}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.2}Improved Mean Flow action generation}{3}{subsection.3.2}\protected@file@percent }
\newlabel{sec:imf_action_generation}{{3.2}{3}{Improved Mean Flow action generation}{subsection.3.2}{}}
\newlabel{sec:imf_action_generation@cref}{{[subsection][2][3]3.2}{[1][3][]3}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.3}Attention Residual policy transformer}{4}{subsection.3.3}\protected@file@percent }
\newlabel{sec:attnres_policy}{{3.3}{4}{Attention Residual policy transformer}{subsection.3.3}{}}
\newlabel{sec:attnres_policy@cref}{{[subsection][3][3]3.3}{[1][3][]4}{}{}{}}
\newlabel{eq:standard_residual_sum}{{10}{4}{Attention Residual policy transformer}{equation.10}{}}
\newlabel{eq:standard_residual_sum@cref}{{[equation][10][]10}{[1][4][]4}{}{}{}}
\newlabel{eq:weighted_residual_sum}{{12}{4}{Attention Residual policy transformer}{equation.12}{}}
\newlabel{eq:weighted_residual_sum@cref}{{[equation][12][]12}{[1][4][]4}{}{}{}}
\newlabel{eq:attnres_weights}{{14}{4}{Attention Residual policy transformer}{equation.14}{}}
\newlabel{eq:attnres_weights@cref}{{[equation][14][]14}{[1][4][]4}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {3.4}Inference and control loop}{4}{subsection.3.4}\protected@file@percent }
\newlabel{sec:inference_loop}{{3.4}{4}{Inference and control loop}{subsection.3.4}{}}
\newlabel{sec:inference_loop@cref}{{[subsection][4][3]3.4}{[1][4][]4}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {1}{\ignorespaces Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image.}}{5}{figure.1}\protected@file@percent }
\newlabel{fig:method_overview}{{1}{5}{Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image}{figure.1}{}}
\newlabel{fig:method_overview@cref}{{[figure][1][]1}{[1][4][]5}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {2}{\ignorespaces Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering.}}{5}{figure.2}\protected@file@percent }
\newlabel{fig:imf_attnres_components}{{2}{5}{Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering}{figure.2}{}}
\newlabel{fig:imf_attnres_components@cref}{{[figure][2][]2}{[1][4][]5}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {section}{\numberline {4}Experiments}{5}{section.4}\protected@file@percent }
\newlabel{sec:experiments}{{4}{5}{Experiments}{section.4}{}}
\newlabel{sec:experiments@cref}{{[section][4][]4}{[1][4][]5}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.1}Setup and metrics}{5}{subsection.4.1}\protected@file@percent }
\newlabel{sec:setup_metrics}{{4.1}{5}{Setup and metrics}{subsection.4.1}{}}
\newlabel{sec:setup_metrics@cref}{{[subsection][1][4]4.1}{[1][4][]5}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.2}Socket peg insertion}{5}{subsection.4.2}\protected@file@percent }
\newlabel{sec:socket_results}{{4.2}{5}{Socket peg insertion}{subsection.4.2}{}}
\newlabel{sec:socket_results@cref}{{[subsection][2][4]4.2}{[1][5][]5}{}{}{}}
\@writefile{lot}{\contentsline {table}{\numberline {1}{\ignorespaces Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt {max\_reward > 4} count.}}{6}{table.1}\protected@file@percent }
\newlabel{tab:socket_results}{{1}{6}{Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt {max\_reward > 4} count}{table.1}{}}
\newlabel{tab:socket_results@cref}{{[table][1][]1}{[1][5][]6}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {3}{\ignorespaces Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image.}}{6}{figure.3}\protected@file@percent }
\newlabel{fig:socket_reward_latency}{{3}{6}{Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image}{figure.3}{}}
\newlabel{fig:socket_reward_latency@cref}{{[figure][3][]3}{[1][5][]6}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.3}Object transfer and ablations}{6}{subsection.4.3}\protected@file@percent }
\newlabel{sec:sim_transfer_results}{{4.3}{6}{Object transfer and ablations}{subsection.4.3}{}}
\newlabel{sec:sim_transfer_results@cref}{{[subsection][3][4]4.3}{[1][6][]6}{}{}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.4}Derived speed-quality comparisons}{6}{subsection.4.4}\protected@file@percent }
\newlabel{sec:derived_comparisons}{{4.4}{6}{Derived speed-quality comparisons}{subsection.4.4}{}}
\newlabel{sec:derived_comparisons@cref}{{[subsection][4][4]4.4}{[1][6][]6}{}{}{}}
\@writefile{toc}{\contentsline {section}{\numberline {5}Limitations}{6}{section.5}\protected@file@percent }
\newlabel{sec:limitations}{{5}{6}{Limitations}{section.5}{}}
\newlabel{sec:limitations@cref}{{[section][5][]5}{[1][6][]6}{}{}{}}
\@writefile{lot}{\contentsline {table}{\numberline {2}{\ignorespaces Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt {max\_reward >= 4} count.}}{7}{table.2}\protected@file@percent }
\newlabel{tab:sim_transfer_results}{{2}{7}{Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt {max\_reward >= 4} count}{table.2}{}}
\newlabel{tab:sim_transfer_results@cref}{{[table][2][]2}{[1][6][]7}{}{}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {4}{\ignorespaces Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering.}}{7}{figure.4}\protected@file@percent }
\newlabel{fig:sim_transfer_ablation}{{4}{7}{Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering}{figure.4}{}}
\newlabel{fig:sim_transfer_ablation@cref}{{[figure][4][]4}{[1][6][]7}{}{}{}}
\def\@LN@column{1}
\@writefile{toc}{\contentsline {section}{\numberline {6}Conclusion}{7}{section.6}\protected@file@percent }
\newlabel{sec:conclusion}{{6}{7}{Conclusion}{section.6}{}}
\newlabel{sec:conclusion@cref}{{[section][6][]6}{[1][7][]7}{}{}{}}
\@writefile{lot}{\contentsline {table}{\numberline {3}{\ignorespaces Derived comparisons from the 100-rollout logs.}}{8}{table.3}\protected@file@percent }
\newlabel{tab:derived_comparisons}{{3}{8}{Derived comparisons from the 100-rollout logs}{table.3}{}}
\newlabel{tab:derived_comparisons@cref}{{[table][3][]3}{[1][6][]8}{}{}{}}
\bibdata{refs}
\bibcite{black2024pi0}{{1}{2024}{{Black et~al.}}{{Black, Brown, Driess, Esmail, Equi, Finn, Fusai, Groom, Hausman, Ichter, et~al.}}}
\bibcite{chi2023diffusion}{{2}{2023}{{Chi et~al.}}{{Chi, Xu, Feng, Cousineau, Du, Burchfiel, Tedrake, and Song}}}
\bibcite{ho2020denoising}{{3}{2020}{{Ho et~al.}}{{Ho, Jain, and Abbeel}}}
\bibcite{peebles2023scalable}{{4}{2023}{{Peebles and Xie}}{{}}}
\bibcite{lipman2023flow}{{5}{2023}{{Lipman et~al.}}{{Lipman, Chen, Ben-Hamu, Nickel, and Le}}}
\bibcite{liu2022flow}{{6}{2022}{{Liu et~al.}}{{Liu, Gong, and Liu}}}
\bibcite{geng2025mean}{{7}{2025{}}{{Geng et~al.}}{{Geng, Deng, Bai, Kolter, and He}}}
\bibcite{geng2025improved}{{8}{2025{}}{{Geng et~al.}}{{Geng, Lu, Wu, Shechtman, Kolter, and He}}}
\bibcite{he2016deep}{{9}{2016}{{He et~al.}}{{He, Zhang, Ren, and Sun}}}
\bibcite{vaswani2017attention}{{10}{2017}{{Vaswani et~al.}}{{Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez, Kaiser, and Polosukhin}}}
\bibcite{kimi2026attention}{{11}{2026}{{Kimi Team}}{{}}}
\bibcite{zhao2023learning}{{12}{2023}{{Zhao et~al.}}{{Zhao, Kumar, Levine, and Finn}}}
\bibcite{shukor2025smolvla}{{13}{2025}{{Shukor et~al.}}{{Shukor, Aubakirova, Capuano, Kooijmans, Palma, Zouitine, Aractingi, Pascal, Russi, Marafioti, et~al.}}}
\bibcite{brohan2022rt}{{14}{2022}{{Brohan et~al.}}{{Brohan, Brown, Carbajal, Chebotar, Dabis, Finn, Gopalakrishnan, Hausman, Herzog, Hsu, et~al.}}}
\bibcite{brohan2023rt}{{15}{2023}{{Brohan et~al.}}{{Brohan, Brown, Carbajal, Chebotar, Chen, Choromanski, Ding, Driess, Dubey, Finn, et~al.}}}
\bibcite{kim2024openvla}{{16}{2024}{{Kim et~al.}}{{Kim, Pertsch, Karamcheti, Xiao, Balakrishna, Nair, Rafailov, Foster, Lam, Sanketi, et~al.}}}
\bibcite{pertsch2025fast}{{17}{2025}{{Pertsch et~al.}}{{Pertsch, Stachowicz, Ichter, Driess, Nair, Vuong, Mees, Finn, and Levine}}}
\def\@LN@column{1}
\gdef \@abspage@last{9}
+119
View File
@@ -0,0 +1,119 @@
\begin{thebibliography}{17}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
\providecommand{\doi}[1]{doi: #1}\else
\providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi
\bibitem[Black et~al.(2024)Black, Brown, Driess, Esmail, Equi, Finn, Fusai,
Groom, Hausman, Ichter, et~al.]{black2024pi0}
K.~Black, N.~Brown, D.~Driess, A.~Esmail, M.~Equi, C.~Finn, N.~Fusai, L.~Groom,
K.~Hausman, B.~Ichter, et~al.
\newblock {$\pi_0$}: A vision-language-action flow model for general robot
control.
\newblock \emph{arXiv preprint arXiv:2410.24164}, 2024.
\bibitem[Chi et~al.(2023)Chi, Xu, Feng, Cousineau, Du, Burchfiel, Tedrake, and
Song]{chi2023diffusion}
C.~Chi, Z.~Xu, S.~Feng, E.~Cousineau, Y.~Du, B.~Burchfiel, R.~Tedrake, and
S.~Song.
\newblock Diffusion policy: Visuomotor policy learning via action diffusion.
\newblock \emph{arXiv preprint arXiv:2303.04137}, 2023.
\bibitem[Ho et~al.(2020)Ho, Jain, and Abbeel]{ho2020denoising}
J.~Ho, A.~Jain, and P.~Abbeel.
\newblock Denoising diffusion probabilistic models.
\newblock In \emph{Advances in Neural Information Processing Systems}, 2020.
\bibitem[Peebles and Xie(2023)]{peebles2023scalable}
W.~Peebles and S.~Xie.
\newblock Scalable diffusion models with transformers.
\newblock In \emph{IEEE/CVF International Conference on Computer Vision}, 2023.
\bibitem[Lipman et~al.(2023)Lipman, Chen, Ben-Hamu, Nickel, and
Le]{lipman2023flow}
Y.~Lipman, R.~T.~Q. Chen, H.~Ben-Hamu, M.~Nickel, and M.~Le.
\newblock Flow matching for generative modeling.
\newblock In \emph{International Conference on Learning Representations}, 2023.
\bibitem[Liu et~al.(2022)Liu, Gong, and Liu]{liu2022flow}
X.~Liu, C.~Gong, and Q.~Liu.
\newblock Flow straight and fast: Learning to generate and transfer data with
rectified flow.
\newblock \emph{arXiv preprint arXiv:2209.03003}, 2022.
\bibitem[Geng et~al.(2025{\natexlab{a}})Geng, Deng, Bai, Kolter, and
He]{geng2025mean}
Z.~Geng, M.~Deng, X.~Bai, J.~Z. Kolter, and K.~He.
\newblock Mean flows for one-step generative modeling.
\newblock \emph{arXiv preprint arXiv:2505.13447}, 2025{\natexlab{a}}.
\bibitem[Geng et~al.(2025{\natexlab{b}})Geng, Lu, Wu, Shechtman, Kolter, and
He]{geng2025improved}
Z.~Geng, Y.~Lu, Z.~Wu, E.~Shechtman, J.~Z. Kolter, and K.~He.
\newblock Improved mean flows: On the challenges of fastforward generative
models.
\newblock \emph{arXiv preprint arXiv:2512.02012}, 2025{\natexlab{b}}.
\bibitem[He et~al.(2016)He, Zhang, Ren, and Sun]{he2016deep}
K.~He, X.~Zhang, S.~Ren, and J.~Sun.
\newblock Deep residual learning for image recognition.
\newblock In \emph{IEEE Conference on Computer Vision and Pattern Recognition},
2016.
\bibitem[Vaswani et~al.(2017)Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez,
Kaiser, and Polosukhin]{vaswani2017attention}
A.~Vaswani, N.~Shazeer, N.~Parmar, J.~Uszkoreit, L.~Jones, A.~N. Gomez,
L.~Kaiser, and I.~Polosukhin.
\newblock Attention is all you need.
\newblock In \emph{Advances in Neural Information Processing Systems}, 2017.
\bibitem[{Kimi Team}(2026)]{kimi2026attention}
{Kimi Team}.
\newblock Attention residuals.
\newblock \emph{arXiv preprint arXiv:2603.15031}, 2026.
\bibitem[Zhao et~al.(2023)Zhao, Kumar, Levine, and Finn]{zhao2023learning}
T.~Z. Zhao, V.~Kumar, S.~Levine, and C.~Finn.
\newblock Learning fine-grained bimanual manipulation with low-cost hardware.
\newblock \emph{arXiv preprint arXiv:2304.13705}, 2023.
\bibitem[Shukor et~al.(2025)Shukor, Aubakirova, Capuano, Kooijmans, Palma,
Zouitine, Aractingi, Pascal, Russi, Marafioti, et~al.]{shukor2025smolvla}
M.~Shukor, D.~Aubakirova, F.~Capuano, P.~Kooijmans, S.~Palma, A.~Zouitine,
M.~Aractingi, C.~Pascal, M.~Russi, A.~Marafioti, et~al.
\newblock Smolvla: A vision-language-action model for affordable and efficient
robotics.
\newblock \emph{arXiv preprint arXiv:2506.01844}, 2025.
\bibitem[Brohan et~al.(2022)Brohan, Brown, Carbajal, Chebotar, Dabis, Finn,
Gopalakrishnan, Hausman, Herzog, Hsu, et~al.]{brohan2022rt}
A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, J.~Dabis, C.~Finn,
K.~Gopalakrishnan, K.~Hausman, A.~Herzog, J.~Hsu, et~al.
\newblock Rt-1: Robotics transformer for real-world control at scale.
\newblock \emph{arXiv preprint arXiv:2212.06817}, 2022.
\bibitem[Brohan et~al.(2023)Brohan, Brown, Carbajal, Chebotar, Chen,
Choromanski, Ding, Driess, Dubey, Finn, et~al.]{brohan2023rt}
A.~Brohan, N.~Brown, J.~Carbajal, Y.~Chebotar, X.~Chen, K.~Choromanski,
T.~Ding, D.~Driess, A.~Dubey, C.~Finn, et~al.
\newblock Rt-2: Vision-language-action models transfer web knowledge to robotic
control.
\newblock In \emph{Conference on Robot Learning}, 2023.
\bibitem[Kim et~al.(2024)Kim, Pertsch, Karamcheti, Xiao, Balakrishna, Nair,
Rafailov, Foster, Lam, Sanketi, et~al.]{kim2024openvla}
M.~J. Kim, K.~Pertsch, S.~Karamcheti, T.~Xiao, A.~Balakrishna, S.~Nair,
R.~Rafailov, E.~Foster, G.~Lam, P.~Sanketi, et~al.
\newblock Openvla: An open-source vision-language-action model.
\newblock \emph{arXiv preprint arXiv:2406.09246}, 2024.
\bibitem[Pertsch et~al.(2025)Pertsch, Stachowicz, Ichter, Driess, Nair, Vuong,
Mees, Finn, and Levine]{pertsch2025fast}
K.~Pertsch, K.~Stachowicz, B.~Ichter, D.~Driess, S.~Nair, Q.~Vuong, O.~Mees,
C.~Finn, and S.~Levine.
\newblock Fast: Efficient action tokenization for vision-language-action
models.
\newblock \emph{arXiv preprint arXiv:2501.09747}, 2025.
\end{thebibliography}
+46
View File
@@ -0,0 +1,46 @@
This is BibTeX, Version 0.99e (TeX Live 2026/Arch Linux)
Capacity: max_strings=200000, hash_size=200000, hash_prime=170003
The top-level auxiliary file: paper.aux
The style file: corlabbrvnat.bst
Database file #1: refs.bib
You've used 17 entries,
2773 wiz_defined-function locations,
667 strings with 8328 characters,
and the built_in function-call counts, 10273 in all, are:
= -- 758
> -- 1061
< -- 2
+ -- 357
- -- 339
* -- 940
:= -- 1707
add.period$ -- 51
call.type$ -- 17
change.case$ -- 159
chr.to.int$ -- 16
cite$ -- 34
duplicate$ -- 362
empty$ -- 586
format.name$ -- 358
if$ -- 2034
int.to.chr$ -- 2
int.to.str$ -- 1
missing$ -- 17
newline$ -- 93
num.names$ -- 68
pop$ -- 343
preamble$ -- 1
purify$ -- 142
quote$ -- 0
skip$ -- 317
stack$ -- 0
substring$ -- 34
swap$ -- 25
text.length$ -- 0
text.prefix$ -- 0
top$ -- 0
type$ -- 187
warning$ -- 0
while$ -- 51
width$ -- 0
write$ -- 211
+205
View File
@@ -0,0 +1,205 @@
# Fdb version 4
["bibtex paper"] 1778915961.47314 "paper.aux" "paper.bbl" "paper" 1778915962.00077 0
"./corlabbrvnat.bst" 1778915960.91983 26694 1972207a84216683d105ea25982a4f25 ""
"./refs.bib" 1778915960.91838 5693 df14ac265e44528611e8ea03e1f31a1e ""
"paper.aux" 1778915961.88858 10768 2556dc9b99b49df6abac644723c92193 "pdflatex"
(generated)
"paper.bbl"
"paper.blg"
(rewritten before read)
["pdflatex"] 1778915961.5118 "paper.tex" "paper.pdf" "paper" 1778915962.00086 0
"/usr/share/texmf-dist/fonts/enc/dvips/base/8r.enc" 1775415801 4850 80dc9bab7f31fb78a000ccfed0e27cab ""
"/usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-t1.enc" 1775415801 2971 def0b6c1f0b107b3b936def894055589 ""
"/usr/share/texmf-dist/fonts/map/fontname/texfonts.map" 1775415801 3524 cb3e574dea2d1052e39280babc910dc8 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8r.tfm" 1775415801 4712 9ef4d7d106579d4b136e1529e1a4533c ""
"/usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8t.tfm" 1775415801 7040 b2bd27e2bfe6f6948cbc3239cae7444f ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm" 1775415801 4524 6bce29db5bc272ba5f332261583fee9c ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm" 1775415801 6880 f19b8995b61c334d78fc734065f6b4d4 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8r.tfm" 1775415801 4408 25b74d011a4c66b7f212c0cc3c90061b ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm" 1775415801 6672 e3ab9e37e925f3045c9005e6d1473d56 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8r.tfm" 1775415801 4640 532ca3305aad10cc01d769f3f91f1029 ""
"/usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8t.tfm" 1775415801 6944 94c55ad86e6ea2826f78ba2240d50df9 ""
"/usr/share/texmf-dist/fonts/tfm/jknappen/ec/ectt1000.tfm" 1775415801 1536 06717a2b50de47d4087ac0e6cd759455 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm" 1775415801 1004 54797486969f23fa377b128694d548df ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex9.tfm" 1775415801 996 a18840b13b499c08ac2de96a99eda4bc ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam10.tfm" 1775415801 916 f87d7c45f9c908e672703b83b72241a3 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam5.tfm" 1775415801 924 9904cf1d39e9767e7a3622f2a125a565 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam7.tfm" 1775415801 928 2dc8d444221b7a635bb58038579b861a ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm10.tfm" 1775415801 908 2921f8a10601f252058503cc6570e581 ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm5.tfm" 1775415801 940 75ac932a52f80982a9f8ea75d03a34cf ""
"/usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm7.tfm" 1775415801 940 228d6584342e91276bf566bcf9716b83 ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmmi6.tfm" 1775415801 1512 f21f83efb36853c0b70002322c1ab3ad ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmmi9.tfm" 1775415801 1524 d89e2d087a9828407a196f428428ef4a ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmr6.tfm" 1775415801 1300 b62933e007d01cfd073f79b963c01526 ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmr9.tfm" 1775415801 1292 6b21b9c2c7bebb38aa2273f7ca0fb3af ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmsy6.tfm" 1775415801 1116 933a60c408fc0a863a92debe84b2d294 ""
"/usr/share/texmf-dist/fonts/tfm/public/cm/cmsy9.tfm" 1775415801 1116 25a7bf822c58caf309a702ef79f4afbb ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmex10.pfb" 1775415801 30251 6afa5cb1d0204815a708a080681d4674 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi10.pfb" 1775415801 36299 5f9df58c2139e7edcf37c8fca4bd384d ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi6.pfb" 1775415801 37166 8ab3487cbe3ab49ebce74c29ea2418db ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi7.pfb" 1775415801 36281 c355509802a035cadc5f15869451dcee ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi9.pfb" 1775415801 36094 798f80770b3b148ceedd006d487db67c ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr10.pfb" 1775415801 35752 024fb6c41858982481f6968b5fc26508 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr6.pfb" 1775415801 32734 69e00a6b65cedb993666e42eedb3d48f ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr7.pfb" 1775415801 32762 224316ccc9ad3ca0423a14971cfa7fc1 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr9.pfb" 1775415801 33993 9b89b85fd2d9df0482bd47194d1d3bf3 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy10.pfb" 1775415801 32569 5e5ddc8df908dea60932f3c484a54c0d ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy7.pfb" 1775415801 32716 08e384dc442464e7285e891af9f45947 ""
"/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy9.pfb" 1775415801 32442 c975af247b6702f7ca0c299af3616b80 ""
"/usr/share/texmf-dist/fonts/type1/public/cm-super/sftt1000.pfb" 1775415801 169201 9ebf99020dde51a5086e186761a34e8f ""
"/usr/share/texmf-dist/fonts/type1/urw/helvetic/uhvr8a.pfb" 1775415801 44648 23115b2a545ebfe2c526c3ca99db8b95 ""
"/usr/share/texmf-dist/fonts/type1/urw/times/utmb8a.pfb" 1775415801 44729 811d6c62865936705a31c797a1d5dada ""
"/usr/share/texmf-dist/fonts/type1/urw/times/utmr8a.pfb" 1775415801 46026 6dab18b61c907687b520c72847215a68 ""
"/usr/share/texmf-dist/fonts/type1/urw/times/utmri8a.pfb" 1775415801 45458 a3faba884469519614ca56ba5f6b1de1 ""
"/usr/share/texmf-dist/fonts/vf/adobe/helvetic/phvr8t.vf" 1775415801 2344 44ff28c9ef2fc97180cd884f900fee71 ""
"/usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf" 1775415801 2340 df9c920cc5688ebbf16a93f45ce7bdd3 ""
"/usr/share/texmf-dist/fonts/vf/adobe/times/ptmr8t.vf" 1775415801 2348 91706c542228501c410c266421fbe30c ""
"/usr/share/texmf-dist/fonts/vf/adobe/times/ptmri8t.vf" 1775415801 2328 6cd7df782b09b29cfc4d93e55b6b9a59 ""
"/usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii" 1775415801 71627 94eb9990bed73c364d7f53f960cc8c5b ""
"/usr/share/texmf-dist/tex/generic/bigintcalc/bigintcalc.sty" 1775415801 40635 c40361e206be584d448876bba8a64a3b ""
"/usr/share/texmf-dist/tex/generic/bitset/bitset.sty" 1775415801 33961 6b5c75130e435b2bfdb9f480a09a39f9 ""
"/usr/share/texmf-dist/tex/generic/gettitlestring/gettitlestring.sty" 1775415801 8371 9d55b8bd010bc717624922fb3477d92e ""
"/usr/share/texmf-dist/tex/generic/iftex/iftex.sty" 1775415801 7984 7dbb9280f03c0a315425f1b4f35d43ee ""
"/usr/share/texmf-dist/tex/generic/iftex/ifvtex.sty" 1775415801 1057 525c2192b5febbd8c1f662c9468335bb ""
"/usr/share/texmf-dist/tex/generic/infwarerr/infwarerr.sty" 1775415801 8356 7bbb2c2373aa810be568c29e333da8ed ""
"/usr/share/texmf-dist/tex/generic/intcalc/intcalc.sty" 1775415801 31769 002a487f55041f8e805cfbf6385ffd97 ""
"/usr/share/texmf-dist/tex/generic/kvdefinekeys/kvdefinekeys.sty" 1775415801 5412 d5a2436094cd7be85769db90f29250a6 ""
"/usr/share/texmf-dist/tex/generic/ltxcmds/ltxcmds.sty" 1775415801 17865 1a9bd36b4f98178fa551aca822290953 ""
"/usr/share/texmf-dist/tex/generic/pdfescape/pdfescape.sty" 1775415801 19007 15924f7228aca6c6d184b115f4baa231 ""
"/usr/share/texmf-dist/tex/generic/pdftexcmds/pdftexcmds.sty" 1775415801 20089 80423eac55aa175305d35b49e04fe23b ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex" 1775415801 1016 1c2b89187d12a2768764b83b4945667c ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorearrows.code.tex" 1775415801 43906 06058dc09064474303f3b5dd62d982c0 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreexternal.code.tex" 1775415801 19324 f4e4c6403dd0f1605fd20ed22fa79dea ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoregraphicstate.code.tex" 1775415801 6038 ccb406740cc3f03bbfb58ad504fe8c27 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreimage.code.tex" 1775415801 6911 f6d4cf5a3fef5cc879d668b810e82868 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorelayers.code.tex" 1775415801 4883 42daaf41e27c3735286e23e48d2d7af9 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreobjects.code.tex" 1775415801 2544 8c06d2a7f0f469616ac9e13db6d2f842 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathconstruct.code.tex" 1775415801 44195 5e390c414de027626ca5e2df888fa68d ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathprocessing.code.tex" 1775415801 17311 e001219836e75b16c4af9a112785f30a ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathusage.code.tex" 1775415801 21302 788a79944eb22192a4929e46963a3067 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepatterns.code.tex" 1775415801 9691 3d42d89522f4650c2f3dc616ca2b925e ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepoints.code.tex" 1775415801 33335 dd1fa4814d4e51f18be97d88bf0da60c ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorequick.code.tex" 1775415801 2965 4c2b1f4e0826925746439038172e5d6f ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorerdf.code.tex" 1775415801 5196 2cc249e0ee7e03da5f5f6589257b1e5b ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorescopes.code.tex" 1775415801 20821 7579108c1e9363e61a0b1584778804aa ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreshade.code.tex" 1775415801 35251 5ff5b5b310c5ac882610e0ccc99095e7 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransformations.code.tex" 1775415801 22012 81b34a0aa8fa1a6158cc6220b00e4f10 ""
"/usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransparency.code.tex" 1775415801 8893 e851de2175338fdf7c17f3e091d94618 ""
"/usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/libraries/tikzlibrarytopaths.code.tex" 1775415801 11518 738408f795261b70ce8dd47459171309 ""
"/usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex" 1775415801 186859 0445d9a41a87648b4723e04765409541 ""
"/usr/share/texmf-dist/tex/generic/pgf/libraries/pgflibraryplothandlers.code.tex" 1775415801 32995 ac577023e12c0e4bd8aa420b2e852d1a ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfint.code.tex" 1775415801 3063 8c415c68a0f3394e45cfeca0b65f6ee6 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex" 1775415801 949 cea70942e7b7eddabfb3186befada2e6 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathcalc.code.tex" 1775415801 13272 7777a64fbd07131a37d276b131c17ee2 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfloat.code.tex" 1775415801 104717 9b2393fbf004a0ce7fa688dbce423848 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.base.code.tex" 1775415801 10165 cec5fa73d49da442e56efc2d605ef154 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.basic.code.tex" 1775415801 28178 41c17713108e0795aac6fef3d275fbca ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.code.tex" 1775415801 9649 85779d3d8d573bfd2cd4137ba8202e60 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.comparison.code.tex" 1775415801 3865 ac538ab80c5cf82b345016e474786549 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.integerarithmetics.code.tex" 1775415801 3177 27d85c44fbfe09ff3b2cf2879e3ea434 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.misc.code.tex" 1775415801 11024 0179538121bc2dba172013a3ef89519f ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.random.code.tex" 1775415801 7889 d0e193914ddc35444510f5b569e26b3d ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.round.code.tex" 1775415801 3379 781797a101f647bab82741a99944a229 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.trigonometric.code.tex" 1775415801 92405 f515f31275db273f97b9d8f52e1b0736 ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathparser.code.tex" 1775415801 37733 0fe471ac50324723cf6ab693e5c0916c ""
"/usr/share/texmf-dist/tex/generic/pgf/math/pgfmathutil.code.tex" 1775415801 8471 c2883569d03f69e8e1cabfef4999cfd7 ""
"/usr/share/texmf-dist/tex/generic/pgf/modules/pgfmodulematrix.code.tex" 1775415801 21211 1e73ec76bd73964d84197cc3d2685b01 ""
"/usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleplot.code.tex" 1775415801 16218 98503859deba28f16813029fd927ed8e ""
"/usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleshapes.code.tex" 1775415801 44792 c4a5a3feba777682c1d16420f2f01a5b ""
"/usr/share/texmf-dist/tex/generic/pgf/pgf.revision.tex" 1775415801 116 760d50e6a16543bf6edb475635793673 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgf.cfg" 1775415801 926 2963ea0dcf6cc6c0a770b69ec46a477b ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-common-pdf.def" 1775415801 5542 32f75a31ea6c3a7e1148cd6d5e93dbb7 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-pdftex.def" 1775415801 12612 7774ba67bfd72e593c4436c2de6201e3 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex" 1775415801 61355 39904e7552da3800a6838d41440943a5 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex" 1775415801 1896 b8e0ca0ac371d74c0ca05583f6313c91 ""
"/usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex" 1775415801 7778 53c8b5623d80238f6a20aa1df1868e63 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex" 1775415801 24033 d8893a1ec4d1bfa101b172754743d340 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex" 1775415801 39784 414c54e866ebab4b801e2ad81d9b21d8 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeyslibraryfiltered.code.tex" 1775415801 37436 50ba7794827e363eec9ea3467c15c6d7 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex" 1775415801 4385 510565c2f07998c8a0e14f0ec07ff23c ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-common.tex" 1775415801 30029 c49ea8f95207c46731469c614daf4e33 ""
"/usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-latex.def" 1775415801 7067 11553488d1600cac6a0cfca012fca111 ""
"/usr/share/texmf-dist/tex/generic/stringenc/stringenc.sty" 1775415801 21514 b7557edcee22835ef6b03ede1802dad4 ""
"/usr/share/texmf-dist/tex/generic/uniquecounter/uniquecounter.sty" 1775415801 7008 f92eaa0a3872ed622bbf538217cd2ab7 ""
"/usr/share/texmf-dist/tex/latex/amsfonts/amsfonts.sty" 1775415801 5949 3f3fd50a8cc94c3d4cbf4fc66cd3df1c ""
"/usr/share/texmf-dist/tex/latex/amsfonts/amssymb.sty" 1775415801 13829 94730e64147574077f8ecfea9bb69af4 ""
"/usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd" 1775415801 961 6518c6525a34feb5e8250ffa91731cff ""
"/usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd" 1775415801 961 d02606146ba5601b5645f987c92e6193 ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsbsy.sty" 1775415801 2222 27db7d52163edae53881b71ff62e754e ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsgen.sty" 1775415801 4173 1b3e76addfb8afcb47db4811d66e1dc6 ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsmath.sty" 1775415801 88471 b1bb09142edddebd46ba986341b867bd ""
"/usr/share/texmf-dist/tex/latex/amsmath/amsopn.sty" 1775415801 4474 c510a88aa5f51b8c773b50a7ee92befd ""
"/usr/share/texmf-dist/tex/latex/amsmath/amstext.sty" 1775415801 2444 9983e1d0683f102e3b190c64a49313aa ""
"/usr/share/texmf-dist/tex/latex/base/article.cls" 1775415801 20144 b966087dda3b194755eb460d32e2ef75 ""
"/usr/share/texmf-dist/tex/latex/base/fontenc.sty" 1775415801 5275 6f9d359641b36842524cdb97716ab75f ""
"/usr/share/texmf-dist/tex/latex/base/size10.clo" 1775415801 8448 686612a86f0e04f41ea577f5ec7e83d8 ""
"/usr/share/texmf-dist/tex/latex/booktabs/booktabs.sty" 1775415801 6078 f1cb470c9199e7110a27851508ed7a5c ""
"/usr/share/texmf-dist/tex/latex/cleveref/cleveref.sty" 1775415801 329481 7fc6b003158402a4c694bc0a1b729308 ""
"/usr/share/texmf-dist/tex/latex/enumitem/enumitem.sty" 1775415801 52272 63d293bc0d496619edb57585740861a2 ""
"/usr/share/texmf-dist/tex/latex/environ/environ.sty" 1775415801 4378 f429f0da968c278653359293040a8f52 ""
"/usr/share/texmf-dist/tex/latex/epstopdf-pkg/epstopdf-base.sty" 1775415801 13886 d1306dcf79a944f6988e688c1785f9ce ""
"/usr/share/texmf-dist/tex/latex/etoolbox/etoolbox.sty" 1775415801 46885 8953c67ffba03252c6090aa19568b8ba ""
"/usr/share/texmf-dist/tex/latex/geometry/geometry.sty" 1775415801 41601 9cf6c5257b1bc7af01a58859749dd37a ""
"/usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg" 1775415801 1213 620bba36b25224fa9b7e1ccb4ecb76fd ""
"/usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg" 1775415801 1224 978390e9c2234eab29404bc21b268d1e ""
"/usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def" 1775415801 19626 23e2822b9b2b5005f4c549ca98b9334d ""
"/usr/share/texmf-dist/tex/latex/graphics/color.sty" 1775415801 7245 a7e8457a46cda4920df85d975267efb4 ""
"/usr/share/texmf-dist/tex/latex/graphics/graphics.sty" 1775415801 18363 69bb4f5538964bfea50d1e6d89cbe69f ""
"/usr/share/texmf-dist/tex/latex/graphics/graphicx.sty" 1775415801 8118 43b99e52946c33a23f5f43b52d5cc5ec ""
"/usr/share/texmf-dist/tex/latex/graphics/keyval.sty" 1775415801 2671 d9941f4bf4750e9b0603c9a2ec54693b ""
"/usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx" 1775415801 2885 9c645d672ae17285bba324998918efd8 ""
"/usr/share/texmf-dist/tex/latex/graphics/trig.sty" 1775415801 4023 e66acf578d6b564c4670fb57ff336a7a ""
"/usr/share/texmf-dist/tex/latex/grfext/grfext.sty" 1775415801 7133 b94bbacbee6e4fdccdc7f810b2aec370 ""
"/usr/share/texmf-dist/tex/latex/hycolor/hycolor.sty" 1775415801 17914 4c28a13fc3d975e6e81c9bea1d697276 ""
"/usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def" 1775415801 48140 0d317d7fb0c7460a10b7b2713db57305 ""
"/usr/share/texmf-dist/tex/latex/hyperref/hyperref.sty" 1775415801 223349 c7928c099a8656537a829ba316c95536 ""
"/usr/share/texmf-dist/tex/latex/hyperref/nameref.sty" 1775415801 11459 697f11f6c439d25d39d2674b99566af4 ""
"/usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def" 1775415801 14249 b94983bbccc8d5739c16cc91d1fd1c3b ""
"/usr/share/texmf-dist/tex/latex/hyperref/puenc.def" 1775415801 117118 2e3ba580751de5583beacf2e5fee69a9 ""
"/usr/share/texmf-dist/tex/latex/kvoptions/kvoptions.sty" 1775415801 22555 6d8e155cfef6d82c3d5c742fea7c992e ""
"/usr/share/texmf-dist/tex/latex/kvsetkeys/kvsetkeys.sty" 1775415801 13815 760b0c02f691ea230f5359c4e1de23a7 ""
"/usr/share/texmf-dist/tex/latex/l3backend/l3backend-pdftex.def" 1775415801 30662 bfd6e864f4ffc5018b0e2b6260c3181c ""
"/usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg" 1775415801 678 4792914a8f45be57bb98413425e4c7af ""
"/usr/share/texmf-dist/tex/latex/lineno/lineno.sty" 1775415801 155535 edbc920f8c4825a66d507b1de69c9ae0 ""
"/usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def" 1775415801 49656 1c61dbb6f95479ba3d6c0033b53590e2 ""
"/usr/share/texmf-dist/tex/latex/microtype/microtype.cfg" 1775415801 27642 f0cea12315babf4d40608f4eeb5c8458 ""
"/usr/share/texmf-dist/tex/latex/microtype/microtype.sty" 1775415801 102845 043c56602c7d94c8a716319df1af5479 ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg" 1775415801 22906 2122f73c0e7dc828f24240c2422dfe25 ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg" 1775415801 5929 2b35ae0f0fb46984dfffa2bc9d09de5c ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg" 1775415801 5594 992ef5c3f8fd1168bb7101a957333065 ""
"/usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg" 1775415801 12427 a6802929d6bd2a4ba6d4f5db602da2b4 ""
"/usr/share/texmf-dist/tex/latex/multirow/multirow.sty" 1775415801 6696 886c9f3087d0b973ed2c19aa79cb3023 ""
"/usr/share/texmf-dist/tex/latex/natbib/natbib.sty" 1775415801 45456 1c8843383c0bd05870c45fa0ebea6cc2 ""
"/usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgf.sty" 1775415801 1090 bae35ef70b3168089ef166db3e66f5b2 ""
"/usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgfcore.sty" 1775415801 373 00b204b1d7d095b892ad31a7494b0373 ""
"/usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-0-65.sty" 1775415801 21013 f4ff83d25bb56552493b030f27c075ae ""
"/usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-1-18.sty" 1775415801 989 c49c8ae06d96f8b15869da7428047b1e ""
"/usr/share/texmf-dist/tex/latex/pgf/frontendlayer/tikz.sty" 1775415801 339 c2e180022e3afdb99c7d0ea5ce469b7d ""
"/usr/share/texmf-dist/tex/latex/pgf/math/pgfmath.sty" 1775415801 306 c56a323ca5bf9242f54474ced10fca71 ""
"/usr/share/texmf-dist/tex/latex/pgf/systemlayer/pgfsys.sty" 1775415801 443 8c872229db56122037e86bcda49e14f3 ""
"/usr/share/texmf-dist/tex/latex/pgf/utilities/pgffor.sty" 1775415801 348 ee405e64380c11319f0e249fed57e6c5 ""
"/usr/share/texmf-dist/tex/latex/pgf/utilities/pgfkeys.sty" 1775415801 274 5ae372b7df79135d240456a1c6f2cf9a ""
"/usr/share/texmf-dist/tex/latex/pgf/utilities/pgfrcs.sty" 1775415801 325 f9f16d12354225b7dd52a3321f085955 ""
"/usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd" 1775415801 1483 47067fbe7c3ffed1ede7aaa7b8549d7a ""
"/usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd" 1775415801 774 61d7da1e9f9e74989b196d147e623736 ""
"/usr/share/texmf-dist/tex/latex/refcount/refcount.sty" 1775415801 9878 9e94e8fa600d95f9c7731bb21dfb67a4 ""
"/usr/share/texmf-dist/tex/latex/rerunfilecheck/rerunfilecheck.sty" 1775415801 9684 a33a14b82ce60d6e77cb9be689d79ee6 ""
"/usr/share/texmf-dist/tex/latex/tcolorbox/tcolorbox.sty" 1775415801 108907 22d0f1983935b2026bb960e8244facba ""
"/usr/share/texmf-dist/tex/latex/tools/verbatim.sty" 1775415801 7532 26d26e9d8f2ca784270d5da8ec7d102b ""
"/usr/share/texmf-dist/tex/latex/trimspaces/trimspaces.sty" 1775415801 1380 971a51b00a14503ddf754cab24c3f209 ""
"/usr/share/texmf-dist/tex/latex/url/url.sty" 1775415801 12796 8edb7d69a20b857904dd0ea757c14ec9 ""
"/usr/share/texmf-dist/tex/latex/xcolor/xcolor.sty" 1775415801 55384 b454dec21c2d9f45ec0b793f0995b992 ""
"/usr/share/texmf-dist/web2c/texmf.cnf" 1775415801 43569 fd570f2fa160877d211e859f687312ba ""
"/var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map" 1778740648 5398031 9ff4c9df8bd43dc57ad0660436b2232f ""
"/var/lib/texmf/web2c/pdftex/pdflatex.fmt" 1778740615 2328330 42e60d011c56833d8cf99c1c1a8869a1 ""
"corl_2026.sty" 1778915960.91905 14861 2509e8e2d8ee9fa6e5f4021973730f28 ""
"paper.aux" 1778915961.88858 10768 2556dc9b99b49df6abac644723c92193 "pdflatex"
"paper.bbl" 1778915961.50957 5480 640ad8c22e2df46df5b9776f80471b97 "bibtex paper"
"paper.out" 1778915961.88858 2172 1cfee1f9ca74e31f8246b69f1b4736b1 "pdflatex"
"paper.tex" 1778915791.17915 28181 86420d1a4e8e129459e1afa202a64339 ""
(generated)
"paper.aux"
"paper.log"
"paper.out"
"paper.pdf"
(rewritten before read)
+356
View File
@@ -0,0 +1,356 @@
PWD /home/droid/project/roboimi/workspace/final
INPUT /usr/share/texmf-dist/web2c/texmf.cnf
INPUT /var/lib/texmf/web2c/pdftex/pdflatex.fmt
INPUT paper.tex
OUTPUT paper.log
INPUT /usr/share/texmf-dist/tex/latex/base/article.cls
INPUT /usr/share/texmf-dist/tex/latex/base/article.cls
INPUT /usr/share/texmf-dist/tex/latex/base/size10.clo
INPUT /usr/share/texmf-dist/tex/latex/base/size10.clo
INPUT /usr/share/texmf-dist/tex/latex/base/size10.clo
INPUT ./corl_2026.sty
INPUT corl_2026.sty
INPUT /usr/share/texmf-dist/tex/latex/lineno/lineno.sty
INPUT /usr/share/texmf-dist/tex/latex/lineno/lineno.sty
INPUT /usr/share/texmf-dist/tex/latex/etoolbox/etoolbox.sty
INPUT /usr/share/texmf-dist/tex/latex/etoolbox/etoolbox.sty
INPUT /usr/share/texmf-dist/tex/latex/kvoptions/kvoptions.sty
INPUT /usr/share/texmf-dist/tex/latex/kvoptions/kvoptions.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/keyval.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/keyval.sty
INPUT /usr/share/texmf-dist/tex/generic/ltxcmds/ltxcmds.sty
INPUT /usr/share/texmf-dist/tex/generic/ltxcmds/ltxcmds.sty
INPUT /usr/share/texmf-dist/tex/latex/kvsetkeys/kvsetkeys.sty
INPUT /usr/share/texmf-dist/tex/latex/kvsetkeys/kvsetkeys.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/color.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/color.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/graphics-def/pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/latex/natbib/natbib.sty
INPUT /usr/share/texmf-dist/tex/latex/natbib/natbib.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hyperref.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hyperref.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/iftex.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/iftex.sty
INPUT /usr/share/texmf-dist/tex/generic/kvdefinekeys/kvdefinekeys.sty
INPUT /usr/share/texmf-dist/tex/generic/kvdefinekeys/kvdefinekeys.sty
INPUT /usr/share/texmf-dist/tex/generic/pdfescape/pdfescape.sty
INPUT /usr/share/texmf-dist/tex/generic/pdfescape/pdfescape.sty
INPUT /usr/share/texmf-dist/tex/generic/pdftexcmds/pdftexcmds.sty
INPUT /usr/share/texmf-dist/tex/generic/pdftexcmds/pdftexcmds.sty
INPUT /usr/share/texmf-dist/tex/generic/infwarerr/infwarerr.sty
INPUT /usr/share/texmf-dist/tex/generic/infwarerr/infwarerr.sty
INPUT /usr/share/texmf-dist/tex/latex/hycolor/hycolor.sty
INPUT /usr/share/texmf-dist/tex/latex/hycolor/hycolor.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/nameref.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/nameref.sty
INPUT /usr/share/texmf-dist/tex/latex/refcount/refcount.sty
INPUT /usr/share/texmf-dist/tex/latex/refcount/refcount.sty
INPUT /usr/share/texmf-dist/tex/generic/gettitlestring/gettitlestring.sty
INPUT /usr/share/texmf-dist/tex/generic/gettitlestring/gettitlestring.sty
INPUT /usr/share/texmf-dist/tex/generic/stringenc/stringenc.sty
INPUT /usr/share/texmf-dist/tex/generic/stringenc/stringenc.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/pd1enc.def
INPUT /usr/share/texmf-dist/tex/generic/intcalc/intcalc.sty
INPUT /usr/share/texmf-dist/tex/generic/intcalc/intcalc.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/puenc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/puenc.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/puenc.def
INPUT /usr/share/texmf-dist/tex/latex/url/url.sty
INPUT /usr/share/texmf-dist/tex/latex/url/url.sty
INPUT /usr/share/texmf-dist/tex/generic/bitset/bitset.sty
INPUT /usr/share/texmf-dist/tex/generic/bitset/bitset.sty
INPUT /usr/share/texmf-dist/tex/generic/bigintcalc/bigintcalc.sty
INPUT /usr/share/texmf-dist/tex/generic/bigintcalc/bigintcalc.sty
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def
INPUT /usr/share/texmf-dist/tex/latex/hyperref/hpdftex.def
INPUT /usr/share/texmf-dist/tex/latex/rerunfilecheck/rerunfilecheck.sty
INPUT /usr/share/texmf-dist/tex/latex/rerunfilecheck/rerunfilecheck.sty
INPUT /usr/share/texmf-dist/tex/generic/uniquecounter/uniquecounter.sty
INPUT /usr/share/texmf-dist/tex/generic/uniquecounter/uniquecounter.sty
INPUT /usr/share/texmf-dist/tex/latex/geometry/geometry.sty
INPUT /usr/share/texmf-dist/tex/latex/geometry/geometry.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/ifvtex.sty
INPUT /usr/share/texmf-dist/tex/generic/iftex/ifvtex.sty
INPUT /usr/share/texmf-dist/tex/latex/base/fontenc.sty
INPUT /usr/share/texmf-dist/tex/latex/base/fontenc.sty
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1ptm.fd
INPUT /usr/share/texmf-dist/fonts/map/fontname/texfonts.map
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.sty
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.sty
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/microtype.cfg
INPUT /usr/share/texmf-dist/tex/latex/booktabs/booktabs.sty
INPUT /usr/share/texmf-dist/tex/latex/booktabs/booktabs.sty
INPUT /usr/share/texmf-dist/tex/latex/multirow/multirow.sty
INPUT /usr/share/texmf-dist/tex/latex/multirow/multirow.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphicx.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphicx.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphics.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/graphics.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/trig.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics/trig.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/graphics.cfg
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsmath.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsmath.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsopn.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amstext.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amstext.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsgen.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsgen.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsbsy.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsbsy.sty
INPUT /usr/share/texmf-dist/tex/latex/amsmath/amsopn.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amssymb.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amssymb.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amsfonts.sty
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/amsfonts.sty
INPUT /usr/share/texmf-dist/tex/latex/enumitem/enumitem.sty
INPUT /usr/share/texmf-dist/tex/latex/enumitem/enumitem.sty
INPUT /usr/share/texmf-dist/tex/latex/tcolorbox/tcolorbox.sty
INPUT /usr/share/texmf-dist/tex/latex/tcolorbox/tcolorbox.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/frontendlayer/tikz.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/frontendlayer/tikz.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgf.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgf.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfrcs.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfrcs.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-common.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfutil-latex.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfrcs.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/pgf.revision.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/pgf.revision.tex
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgfcore.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/basiclayer/pgfcore.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/systemlayer/pgfsys.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/systemlayer/pgfsys.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeyslibraryfiltered.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgf.cfg
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-pdftex.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-pdftex.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsys-common-pdf.def
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsyssoftpath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/systemlayer/pgfsysprotocol.code.tex
INPUT /usr/share/texmf-dist/tex/latex/xcolor/xcolor.sty
INPUT /usr/share/texmf-dist/tex/latex/xcolor/xcolor.sty
INPUT /usr/share/texmf-dist/tex/latex/graphics-cfg/color.cfg
INPUT /usr/share/texmf-dist/tex/latex/graphics/mathcolor.ltx
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcore.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathutil.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathparser.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.basic.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.trigonometric.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.random.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.comparison.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.base.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.round.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.misc.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfunctions.integerarithmetics.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathcalc.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmathfloat.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfint.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepoints.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathconstruct.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathusage.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorescopes.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoregraphicstate.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransformations.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorequick.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreobjects.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepathprocessing.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorearrows.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreshade.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreimage.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoreexternal.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorelayers.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcoretransparency.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorepatterns.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/basiclayer/pgfcorerdf.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleshapes.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/modules/pgfmoduleplot.code.tex
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-0-65.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-0-65.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-1-18.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/compatibility/pgfcomp-version-1-18.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgffor.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgffor.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfkeys.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/utilities/pgfkeys.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgfkeys.code.tex
INPUT /usr/share/texmf-dist/tex/latex/pgf/math/pgfmath.sty
INPUT /usr/share/texmf-dist/tex/latex/pgf/math/pgfmath.sty
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/math/pgfmath.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/utilities/pgffor.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/tikz.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/libraries/pgflibraryplothandlers.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/libraries/pgflibraryplothandlers.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/modules/pgfmodulematrix.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/libraries/tikzlibrarytopaths.code.tex
INPUT /usr/share/texmf-dist/tex/generic/pgf/frontendlayer/tikz/libraries/tikzlibrarytopaths.code.tex
INPUT /usr/share/texmf-dist/tex/latex/tools/verbatim.sty
INPUT /usr/share/texmf-dist/tex/latex/tools/verbatim.sty
INPUT /usr/share/texmf-dist/tex/latex/environ/environ.sty
INPUT /usr/share/texmf-dist/tex/latex/environ/environ.sty
INPUT /usr/share/texmf-dist/tex/latex/trimspaces/trimspaces.sty
INPUT /usr/share/texmf-dist/tex/latex/trimspaces/trimspaces.sty
INPUT /usr/share/texmf-dist/tex/latex/cleveref/cleveref.sty
INPUT /usr/share/texmf-dist/tex/latex/cleveref/cleveref.sty
INPUT /usr/share/texmf-dist/tex/latex/l3backend/l3backend-pdftex.def
INPUT /usr/share/texmf-dist/tex/latex/l3backend/l3backend-pdftex.def
INPUT ./paper.aux
INPUT ./paper.aux
INPUT paper.aux
OUTPUT paper.aux
INPUT /usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii
INPUT /usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii
INPUT /usr/share/texmf-dist/tex/context/base/mkii/supp-pdf.mkii
INPUT /usr/share/texmf-dist/tex/latex/epstopdf-pkg/epstopdf-base.sty
INPUT /usr/share/texmf-dist/tex/latex/epstopdf-pkg/epstopdf-base.sty
INPUT /usr/share/texmf-dist/tex/latex/grfext/grfext.sty
INPUT /usr/share/texmf-dist/tex/latex/grfext/grfext.sty
INPUT /usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg
INPUT /usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg
INPUT /usr/share/texmf-dist/tex/latex/latexconfig/epstopdf-sys.cfg
INPUT ./paper.out
INPUT ./paper.out
INPUT paper.out
INPUT paper.out
OUTPUT paper.pdf
INPUT ./paper.out
INPUT ./paper.out
OUTPUT paper.out
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-ptm.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-cmr.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsa.fd
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam10.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msa.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam5.tfm
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd
INPUT /usr/share/texmf-dist/tex/latex/amsfonts/umsb.fd
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm10.tfm
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg
INPUT /usr/share/texmf-dist/tex/latex/microtype/mt-msb.cfg
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm5.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/jknappen/ec/ectt1000.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd
INPUT /usr/share/texmf-dist/tex/latex/psnfss/t1phv.fd
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8t.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT /var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map
INPUT /usr/share/texmf-dist/fonts/enc/dvips/base/8r.enc
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmr8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8r.tfm
INPUT /usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-t1.enc
INPUT /usr/share/texmf-dist/fonts/vf/adobe/helvetic/phvr8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/helvetic/phvr8r.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmr8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmr8r.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmr9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmr6.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmmi9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmmi6.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmsy9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/cm/cmsy6.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex9.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/cmextra/cmex7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam10.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msam7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm10.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/public/amsfonts/symbols/msbm7.tfm
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8t.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmb8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmb8r.tfm
INPUT ./paper.bbl
INPUT ./paper.bbl
INPUT paper.bbl
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8t.tfm
INPUT /usr/share/texmf-dist/fonts/vf/adobe/times/ptmri8t.vf
INPUT /usr/share/texmf-dist/fonts/tfm/adobe/times/ptmri8r.tfm
INPUT paper.aux
INPUT ./paper.out
INPUT ./paper.out
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmex10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi6.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi7.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmmi9.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr6.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr7.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmr9.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy10.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy7.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmsy9.pfb
INPUT /usr/share/texmf-dist/fonts/type1/public/cm-super/sftt1000.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/helvetic/uhvr8a.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/times/utmb8a.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/times/utmr8a.pfb
INPUT /usr/share/texmf-dist/fonts/type1/urw/times/utmri8a.pfb
+14
View File
@@ -0,0 +1,14 @@
\BOOKMARK [1][-]{section.1}{\376\377\000I\000n\000t\000r\000o\000d\000u\000c\000t\000i\000o\000n}{}% 1
\BOOKMARK [1][-]{section.2}{\376\377\000R\000e\000l\000a\000t\000e\000d\000\040\000W\000o\000r\000k}{}% 2
\BOOKMARK [1][-]{section.3}{\376\377\000M\000e\000t\000h\000o\000d}{}% 3
\BOOKMARK [2][-]{subsection.3.1}{\376\377\000P\000o\000l\000i\000c\000y\000\040\000f\000o\000r\000m\000u\000l\000a\000t\000i\000o\000n}{section.3}% 4
\BOOKMARK [2][-]{subsection.3.2}{\376\377\000I\000m\000p\000r\000o\000v\000e\000d\000\040\000M\000e\000a\000n\000\040\000F\000l\000o\000w\000\040\000a\000c\000t\000i\000o\000n\000\040\000g\000e\000n\000e\000r\000a\000t\000i\000o\000n}{section.3}% 5
\BOOKMARK [2][-]{subsection.3.3}{\376\377\000A\000t\000t\000e\000n\000t\000i\000o\000n\000\040\000R\000e\000s\000i\000d\000u\000a\000l\000\040\000p\000o\000l\000i\000c\000y\000\040\000t\000r\000a\000n\000s\000f\000o\000r\000m\000e\000r}{section.3}% 6
\BOOKMARK [2][-]{subsection.3.4}{\376\377\000I\000n\000f\000e\000r\000e\000n\000c\000e\000\040\000a\000n\000d\000\040\000c\000o\000n\000t\000r\000o\000l\000\040\000l\000o\000o\000p}{section.3}% 7
\BOOKMARK [1][-]{section.4}{\376\377\000E\000x\000p\000e\000r\000i\000m\000e\000n\000t\000s}{}% 8
\BOOKMARK [2][-]{subsection.4.1}{\376\377\000S\000e\000t\000u\000p\000\040\000a\000n\000d\000\040\000m\000e\000t\000r\000i\000c\000s}{section.4}% 9
\BOOKMARK [2][-]{subsection.4.2}{\376\377\000S\000o\000c\000k\000e\000t\000\040\000p\000e\000g\000\040\000i\000n\000s\000e\000r\000t\000i\000o\000n}{section.4}% 10
\BOOKMARK [2][-]{subsection.4.3}{\376\377\000O\000b\000j\000e\000c\000t\000\040\000t\000r\000a\000n\000s\000f\000e\000r\000\040\000a\000n\000d\000\040\000a\000b\000l\000a\000t\000i\000o\000n\000s}{section.4}% 11
\BOOKMARK [2][-]{subsection.4.4}{\376\377\000D\000e\000r\000i\000v\000e\000d\000\040\000s\000p\000e\000e\000d\000-\000q\000u\000a\000l\000i\000t\000y\000\040\000c\000o\000m\000p\000a\000r\000i\000s\000o\000n\000s}{section.4}% 12
\BOOKMARK [1][-]{section.5}{\376\377\000L\000i\000m\000i\000t\000a\000t\000i\000o\000n\000s}{}% 13
\BOOKMARK [1][-]{section.6}{\376\377\000C\000o\000n\000c\000l\000u\000s\000i\000o\000n}{}% 14
Binary file not shown.
+306
View File
@@ -0,0 +1,306 @@
\documentclass{article}
\usepackage{corl_2026}
\usepackage[T1]{fontenc}
\usepackage{microtype}
\usepackage{booktabs}
\usepackage{multirow}
\usepackage{graphicx}
\usepackage{amsmath,amssymb}
\usepackage{enumitem}
\usepackage{tcolorbox}
\usepackage[capitalize]{cleveref}
\usepackage{url}
\title{iMF-AttnRes: Fast Mean-Flow Action Generation with Attention Residuals for Simulated Robot Manipulation}
\author{Anonymous Authors}
\begin{document}
\maketitle
\begin{abstract}
Diffusion-style vision-language-action policies provide expressive action distributions, but iterative inference can be too slow for closed-loop manipulation. We study whether improved Mean Flow (iMF), originally motivated by fast-forward generative modeling, can be adapted to action generation so that a policy uses one to a few flow evaluations rather than long denoising chains. We combine iMF with Attention Residuals (AttnRes), which replace fixed residual accumulation in a deep policy transformer with learned depth-wise aggregation over previous layer outputs. In RoboIMI socket peg insertion, the best iMF-AttnRes policy reaches an average reward of 1513.56 and median reward of 1901.5 with 15.122 ms average inference time, compared with diffusion-style infer100 baselines at 861.53/338.291 ms and 1019.39/397.912 ms. In RoboIMI object transfer, the best iMF-AttnRes policy reaches average reward 526.22 and 44/100 success-like episodes, compared with a native diffusion-policy baseline at 319.2 and 29/100. These results suggest that average-flow action generation is a promising route to low-latency robot policies, while also revealing sensitivity to horizon choices, vision-token design, and simulation-to-real validation.
\end{abstract}
\keywords{Robot learning, imitation learning, flow matching, vision-language-action models}
\section{Introduction}
\label{sec:introduction}
Learning-based robot manipulation policies increasingly use expressive generative action models. Diffusion Policy showed that action diffusion can be an effective visuomotor policy class for manipulation~\citep{black2024pi0,chi2023diffusion}, inheriting the representational flexibility of denoising diffusion models~\citep{ho2020denoising} and transformer-based diffusion backbones~\citep{peebles2023scalable}. However, this expressivity often comes with a deployment cost: the policy must be evaluated repeatedly during sampling. In the RoboIMI socket peg experiments studied here, two diffusion-style VLA baselines use infer100 sampling and require 338.291 ms and 397.912 ms per inference on their recorded rollouts. Such latencies are undesirable when the robot must repeatedly close the perception-action loop.
This paper asks a direct question: can a robot action generator retain the quality benefits of generative policies while reducing inference to one or a few evaluations? We explore this question by adapting improved Mean Flow (iMF) to VLA-style imitation learning. Flow matching and rectified flow formulate generation through continuous vector fields~\citep{lipman2023flow,liu2022flow}. Mean Flow reframes generation around average velocity, enabling one-step generation~\citep{geng2025mean}; iMF further addresses instability caused by substituting conditional velocities into nonlinear mean-flow relations~\citep{geng2025improved}. For robot control, this distinction matters because fewer evaluations directly increase control responsiveness.
We pair iMF with Attention Residuals (AttnRes). Residual connections are essential for deep visual and transformer networks~\citep{he2016deep,vaswani2017attention}, but standard residual accumulation adds all layer outputs with fixed unit weight. The motivation in our notes is to rewrite residual networks as a sum over layer increments, then replace the uniform sum with a learned softmax aggregation. AttnRes implements this view by letting each layer attend over preceding residual outputs~\citep{kimi2026attention}. We use this mechanism in the policy transformer to reduce depth-wise feature dilution while keeping the action generator fast.
Our contributions are:
\begin{enumerate}[leftmargin=*]
\item We formulate an iMF-AttnRes VLA policy for simulated robot imitation learning, combining one-to-few-step average-flow action generation with learned residual aggregation.
\item We provide 100-rollout evaluations on two RoboIMI manipulation environments, socket peg insertion and object transfer, comparing against diffusion-style VLA baselines, a native Diffusion Policy baseline, ACT-style action chunking, and SmolVLA-style compact VLA inference~\citep{zhao2023learning,shukor2025smolvla}.
\item We report both task quality and deployment speed. On socket insertion, the strongest iMF-AttnRes run improves average reward over the ph16 diffusion-style baseline by 1.76$\times$ and inference FPS by 22.97$\times$; on object transfer, the strongest iMF-AttnRes run improves average reward over the native diffusion-policy baseline by 1.65$\times$.
\end{enumerate}
We deliberately keep the claims scoped to simulation: hardware varies across runs, and real-robot transfer remains future work.
\section{Related Work}
\label{sec:related_work}
\paragraph{Diffusion and flow policies for robot action generation.}
Diffusion Policy introduced action diffusion as a visuomotor policy learning framework~\citep{black2024pi0,chi2023diffusion}, building on denoising diffusion objectives~\citep{ho2020denoising}. Diffusion transformers further show that transformer backbones can scale generative modeling~\citep{peebles2023scalable}. These models are expressive because they iteratively refine samples, but iterative refinement is also a latency source. Flow matching provides an alternative continuous generative formulation by learning vector fields that transport probability paths~\citep{lipman2023flow}. Rectified flow emphasizes straighter transport paths and faster sampling~\citep{liu2022flow}. Our method follows this fast-generation lineage but uses the improved mean-flow relation to target one-to-few-step action generation.
\paragraph{Vision-language-action models and action chunking.}
Transformer robot policies such as RT-1 and RT-2 show how large sequence models can map observations and task context to robot actions~\citep{brohan2022rt,brohan2023rt}. OpenVLA studies open VLA modeling at scale~\citep{kim2024openvla}, while compact or efficient VLA models such as SmolVLA and FAST reduce deployment cost through smaller backbones or efficient action tokenization~\citep{shukor2025smolvla,pertsch2025fast}. Action Chunking with Transformers (ACT) predicts chunks of future actions to improve temporal consistency and execution efficiency~\citep{zhao2023learning}. The iMF-AttnRes policy is complementary: it retains chunked execution but changes the generative action sampler so that each chunk can be produced with one to a few evaluations.
\paragraph{Residual aggregation and attention residuals.}
Standard residual networks update $x_t=x_{t-1}+y_t$, so recursively $x_t=y_0+y_1+\cdots+y_t$. Equivalently, the next layer receives a uniform aggregation of all previous residual increments. A natural generalization is
\begin{equation}
y_{t+1}=f_{t+1}\left(\sum_{s=0}^t a_{t+1,s}y_s\right),\qquad a_{t+1,s}\ge0,\quad \sum_{s=0}^t a_{t+1,s}=1.
\end{equation}
AttnRes instantiates the weights with an attention distribution over previous residual outputs,
\begin{equation}
a_{t+1,s}\propto\exp\left(w_{t+1}^{\top}\mathrm{RMSNorm}(y_s)\right).
\end{equation}
This preserves the residual pathway while allowing each layer to select useful earlier representations instead of uniformly accumulating all of them~\citep{kimi2026attention}. We use this mechanism in the policy transformer; our ablations suggest that applying AttnRes too broadly inside the vision stack is not automatically beneficial.
\section{Method}
\label{sec:method}
\subsection{Policy formulation}
\label{sec:policy_formulation}
We consider simulated imitation learning in RoboIMI. At each control step, the policy receives visual observations, robot state, and a task context, and predicts an action chunk of length \texttt{exec}. The main iMF-AttnRes socket and object-transfer policies use ph32/exec16 or related horizon-execution settings. The rollout evaluator records cumulative reward, median reward, mean maximum reward, nonzero-reward episode count, task-specific success-like threshold count, inference FPS, control FPS, and inference latency.
The policy contains three conceptual parts. First, an observation encoder maps visual and state inputs into policy tokens. Second, a transformer policy core uses AttnRes in place of fixed residual accumulation. Third, the action generator uses iMF to predict an average flow over action noise-to-data paths, enabling one-to-few inference evaluations. In the strongest socket variants, the generator uses infer2 or infer3; in the strongest object-transfer run, it uses infer1.
\subsection{Improved Mean Flow action generation}
\label{sec:imf_action_generation}
Classical flow matching trains an instantaneous velocity field. Let $x_t$ denote a sample at time $t$ and let the instantaneous flow satisfy
\begin{equation}
\frac{d x_t}{dt} = v(x_t,t).
\end{equation}
If a policy takes very few integration or denoising steps, using only this local instantaneous velocity can cause large path errors when the learned trajectory is curved. Mean Flow addresses this by training an average velocity over an interval. For two times $r<t$, define
\begin{equation}
\bar{v}(z_t,r,t) = \frac{1}{t-r}\int_r^t v(z_\tau,\tau)d\tau.
\end{equation}
The model receives $(z_t,r,t)$ and predicts $u_\theta(z_t,r,t)$, an approximation to $\bar{v}(z_t,r,t)$. At inference, this average velocity is used to move directly across a larger interval, replacing a long sequence of small denoising or ODE steps.
The useful training identity comes from differentiating $(t-r)\bar{v}$ with respect to $t$:
\begin{equation}
\bar{v}(z_t,r,t)=v(z_t,t) - (t-r)\frac{d}{dt}\bar{v}(z_t,r,t).
\end{equation}
The total derivative contains the dependence of $\bar{v}$ on the current state and time, and can be written as a Jacobian-vector product,
\begin{equation}
\frac{d}{dt}\bar{v}(z_t,r,t)
= \mathrm{jvp}\big(\bar{v},(z,r,t),(v,0,1)\big).
\end{equation}
Mean Flow uses this relation to train an average-velocity predictor for one-step generation~\citep{geng2025mean}. In our setting, the target action sample and noise sample provide a conditional straight-path velocity, but the nonlinear JVP expression should conceptually depend on the marginal velocity field. Directly substituting the conditional velocity inside the nonlinear term can destabilize training. We therefore use the improved Mean Flow correction~\citep{geng2025improved}:
\begin{equation}
V_\theta(z_t,r,t) = u_\theta(z_t,r,t) + (t-r)\mathrm{JVP}_{\mathrm{sg}}\big(u_\theta; v_\theta\big).
\end{equation}
Here $v_\theta$ is obtained from the same network at the instantaneous case, and the stop-gradient JVP direction prevents the nonlinear term from destabilizing the target. The supervised target remains the conditional straight-path velocity available from the imitation action sample and noise sample, but the learned branch used at deployment is the average-flow predictor $u_\theta$. This is the mechanism that allows infer1, infer2, and infer3 policies to compete with infer100 diffusion-style baselines.
\subsection{Attention Residual policy transformer}
\label{sec:attnres_policy}
We spell out AttnRes by first rewriting an ordinary residual stack. Let $x_l$ be the hidden state entering layer $l$, and let the layer output increment be
\begin{equation}
y_l = f_l(x_{l-1}), \qquad x_l = x_{l-1}+y_l .
\end{equation}
With the convention $y_0=x_0$, recursively expanding the residual updates gives
\begin{equation}
x_l = y_0+y_1+\cdots+y_l = \sum_{s=0}^{l} y_s .
\end{equation}
Therefore, the next residual block can be written as
\begin{equation}
y_{l+1}=f_{l+1}(x_l)
= f_{l+1}\left(\sum_{s=0}^{l} y_s\right).
\label{eq:standard_residual_sum}
\end{equation}
This makes the implicit assumption clear: a standard residual network gives every previous increment a fixed coefficient of one before layer $l+1$ computes the next increment. If we normalize the coefficients only to emphasize their relative importance, this is equivalent to uniform depth aggregation,
\begin{equation}
y_{l+1}=f_{l+1}\left((l+1)\sum_{s=0}^{l}\frac{1}{l+1}y_s\right),
\end{equation}
so the layer cannot choose which earlier residual features are most useful.
AttnRes replaces this fixed aggregation with learned, layer-dependent weights. Instead of feeding $f_{l+1}$ the unweighted residual sum, we form
\begin{equation}
\tilde{x}_l = \sum_{s=0}^{l} a_{l+1,s} y_s,
\qquad
a_{l+1,s}\ge 0,
\qquad
\sum_{s=0}^{l} a_{l+1,s}=1,
\label{eq:weighted_residual_sum}
\end{equation}
then compute
\begin{equation}
y_{l+1}=f_{l+1}(\tilde{x}_l),
\qquad
x_{l+1}=x_l+y_{l+1}.
\end{equation}
The attention weights are obtained by scoring each previous residual increment with a learned query vector for the current layer:
\begin{equation}
e_{l+1,s}=w_{l+1}^{\top}\mathrm{RMSNorm}(y_s),
\qquad
a_{l+1,s}=\frac{\exp(e_{l+1,s})}{\sum_{j=0}^{l}\exp(e_{l+1,j})}.
\label{eq:attnres_weights}
\end{equation}
Equations~\eqref{eq:standard_residual_sum}--\eqref{eq:attnres_weights} show the difference in one line: standard residuals use a fixed sum $\sum_s y_s$, whereas AttnRes uses an attention-weighted sum $\sum_s a_{l+1,s}y_s$ whose coefficients depend on the current layer. This keeps the residual pathway but lets each policy layer select earlier visual-state-action features rather than uniformly accumulating all previous increments. In our object-transfer ablations, the best settings apply AttnRes in the policy transformer; replacing residuals throughout the vision encoder underperforms, suggesting that the placement of learned residual aggregation is important.
\subsection{Inference and control loop}
\label{sec:inference_loop}
At deployment, the iMF-AttnRes policy samples an action chunk with one to a few average-flow evaluations and then executes the chunk for \texttt{exec} control steps. This contrasts with diffusion-style baselines whose run names include infer100. For socket insertion, the strongest iMF-AttnRes policies use ph32/exec16 and infer2 or infer3. For object transfer, the strongest iMF-AttnRes run uses ph32/exec16 and infer1. The resulting control loop is simple: encode observations, run the AttnRes transformer, evaluate the iMF action generator a small number of times, execute the predicted chunk, and replan on the next observation window.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-method-overview}}]
\small Paper Banana prompt: create a clean 16:9 academic system diagram for iMF-AttnRes VLA. Show multi-view images, robot state, and optional language/task conditioning entering a VLA encoder; a policy transformer with AttnRes depth-wise residual aggregation; an improved Mean Flow action generator predicting average flow with one-to-few evaluations; and an exec-length action chunk controlling RoboIMI socket-insert and sim\_transfer robots. Use muted conference-paper colors, clear arrows, no decorative 3D, and labels for iMF, AttnRes, infer1/infer2/infer3, and exec16.
\end{tcolorbox}
\caption{Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image.}
\label{fig:method_overview}
\end{figure}
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-imf-attnres-components}}]
\small Paper Banana prompt: create a 16:9 technical diagram contrasting classical flow matching instantaneous velocity $dx_t/dt=v(x_t,t)$; Mean Flow average velocity over $[r,t]$ with the JVP correction term; improved Mean Flow training relation $V_\theta(z_t)=u_\theta(z_t)+(t-r)\mathrm{JVP}_{\mathrm{sg}}(u_\theta;v_\theta)$; and AttnRes weighted residual aggregation $a_{t+1,s}$ proportional to $\exp(w_{t+1}\cdot\mathrm{RMSNorm}(y_s))$. Use equation callouts, minimal arrows, and a final callout saying one-to-few-step action generation for robot control.
\end{tcolorbox}
\caption{Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering.}
\label{fig:imf_attnres_components}
\end{figure}
\section{Experiments}
\label{sec:experiments}
\subsection{Setup and metrics}
\label{sec:setup_metrics}
We evaluate in two RoboIMI simulation environments. The socket peg task measures progress toward inserting a peg-like object into a socket. The sim\_transfer task measures object-transfer manipulation. Each reported row uses 100 rollouts. Metrics are copied directly from the rollout logs: average cumulative reward, median reward, mean maximum reward, maximum cumulative reward, count of nonzero-reward episodes, count of success-like episodes, inference FPS, control FPS, and average inference time when available. For socket insertion, the recorded success-like threshold is \texttt{max\_reward > 4}; for sim\_transfer, it is \texttt{max\_reward >= 4}. Hardware differs across runs, so speed comparisons are practical deployment measurements rather than perfectly normalized throughput benchmarks.
\subsection{Socket peg insertion}
\label{sec:socket_results}
\Cref{tab:socket_results} summarizes the socket peg insertion results. The best iMF-AttnRes policy by average reward is the infer3 model, with average reward 1513.56 and median reward 1901.5. It exceeds both diffusion-style infer100 baselines: the ph16 baseline obtains average reward 861.53 and median reward 668.0, while the ph32/exec16 baseline obtains average reward 1019.39 and median reward 804.0. The latency gap is large. The iMF-AttnRes infer2 and infer3 policies require 14.409 ms and 15.122 ms per inference, while the two infer100 baselines require 338.291 ms and 397.912 ms.
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{3.5pt}
\caption{Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt{max\_reward > 4} count.}
\label{tab:socket_results}
\begin{tabular}{llrrrrrr}
\toprule
Method & Setting & Avg. reward & Median & Avg. max & Nonzero & Success-like & Latency ms \\
\midrule
iMF-AttnRes & infer1, ph32, exec16, 50k & 1275.22 & 1490.5 & 3.12 & 84/100 & 0/100 & 16.186 \\
Diffusion-style VLA & infer100, ph16, 150k & 861.53 & 668.0 & 2.58 & 97/100 & 2/100 & 338.291 \\
Diffusion-style VLA & infer100, ph32, exec16, 150k & 1019.39 & 804.0 & 2.47 & 92/100 & 4/100 & 397.912 \\
iMF-AttnRes & infer2, ph32, exec16, 150k & 1472.62 & 1825.0 & 3.29 & 83/100 & 5/100 & 14.409 \\
iMF-AttnRes & infer3, ph32, exec16, 150k & \textbf{1513.56} & \textbf{1901.5} & 3.28 & 83/100 & 3/100 & 15.122 \\
ACT & action chunking & 289.60 & 13.0 & 1.29 & 55/100 & 1/100 & 100.212 \\
SmolVLA & 100k & 466.16 & 106.0 & 1.72 & 89/100 & 0/100 & \textbf{2.614} \\
\bottomrule
\end{tabular}
\end{table*}
The nonzero-reward count reveals an important caveat. The ph16 diffusion-style baseline has 97/100 nonzero episodes, higher than the iMF-AttnRes infer2 and infer3 policies at 83/100. Yet its average and median rewards are much lower. Thus, in this task, nonzero contact or partial progress is not sufficient; the iMF-AttnRes policies more often produce high-reward trajectories when they engage the task successfully. SmolVLA is the fastest method in raw latency and control FPS, but its reward is lower than iMF-AttnRes. ACT underperforms both iMF-AttnRes and the diffusion-style VLA baselines in this setup.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-socket-reward-latency}}]
\small Paper Banana prompt: create a 4:3 publication-quality reward-latency comparison for socket peg insertion. Plot methods from the socket table with average reward on the y-axis and average inference time or inverse latency on the x-axis. Highlight iMF-AttnRes infer2/infer3 at 1472.62/1513.56 average reward and 14.409/15.122 ms, diffusion-style infer100 baselines at 861.53/1019.39 average reward and 338.291/397.912 ms, ACT at 289.6 and 100.2116 ms, and SmolVLA at 466.16 and 2.614 ms. Use log-scale latency if helpful and label the speed-quality Pareto frontier.
\end{tcolorbox}
\caption{Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image.}
\label{fig:socket_reward_latency}
\end{figure}
\subsection{Object transfer and ablations}
\label{sec:sim_transfer_results}
\Cref{tab:sim_transfer_results} reports the object-transfer results. The strongest iMF-AttnRes run reaches average reward 526.22 and 44/100 success-like episodes, exceeding the native diffusion-policy DiT/DDPM/ResNet baseline at average reward 319.2 and 29/100 success-like episodes. This is the clearest sim\_transfer evidence that iMF-AttnRes can improve both reward and success-like threshold count.
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{3.2pt}
\caption{Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt{max\_reward >= 4} count.}
\label{tab:sim_transfer_results}
\begin{tabular}{llrrrrrr}
\toprule
Method & Setting & Avg. reward & Median & Avg. max & Nonzero & Success-like & Inf. FPS \\
\midrule
Native Diffusion Policy & DiT + DDPM + ResNet & 319.20 & 6.0 & 1.68 & 55/100 & 29/100 & 32.09 \\
Diffusion-style VLA & emb384, layer18 & 233.52 & 0.0 & 1.12 & 39/100 & 17/100 & 1.859 \\
iMF multi-token ResNet18 & step34999, ph16, exec08 & 260.66 & 0.0 & 1.12 & 33/100 & 23/100 & \textbf{441.80} \\
iMF full AttnRes vision & ph16, exec08, 50k & 228.42 & 0.0 & 0.94 & 31/100 & 16/100 & 55.997 \\
iMF-AttnRes DiT only & ph16, exec16, 50k & 240.64 & 0.0 & 1.02 & 32/100 & 19/100 & 137.996 \\
iMF-AttnRes DiT only & ph32, exec08, 50k & 163.28 & 0.0 & 0.76 & 26/100 & 12/100 & 85.638 \\
iMF-AttnRes DiT only & ph32, exec32, 50k & 260.72 & 0.0 & 1.18 & 38/100 & 21/100 & 9.557 \\
iMF-AttnRes DiT only & ph16, exec08, 50k & 229.56 & 0.0 & 1.18 & 41/100 & 18/100 & 69.984 \\
iMF-AttnRes DiT only & ph08, exec08, 50k & 237.02 & 0.0 & 1.32 & 48/100 & 18/100 & 69.192 \\
iMF-AttnRes DiT only & ph32, exec16, 50k & 49.88 & 0.0 & 0.32 & 13/100 & 4/100 & 138.903 \\
iMF-AttnRes & infer1, ph32, exec16, 50k & \textbf{526.22} & \textbf{86.0} & \textbf{2.14} & \textbf{63/100} & \textbf{44/100} & 8.911 \\
\bottomrule
\end{tabular}
\end{table*}
The ablations are mixed and therefore useful. The ResNet18 multi-token iMF model is extremely fast at 441.80 inference FPS, but its average reward is 260.66, below the native diffusion-policy baseline. The full-AttnRes vision model reaches 228.42 average reward, suggesting that replacing residuals inside the vision encoder is not automatically helpful. Horizon and execution length are also sensitive: the ph32/exec16 DiT-only iMF variant reaches only 49.88 average reward despite high inference FPS. The strongest object-transfer run therefore combines iMF-AttnRes with the right horizon and execution setting rather than showing a universally dominant architectural change.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-sim-transfer-ablation}}]
\small Paper Banana prompt: create a 4:3 grouped bar chart for sim\_transfer ablations. Show average reward and success-like episode count for native diffusion policy, best sim-transfer iMF-AttnRes infer1, ResNet18 multi-token iMF, full-AttnRes vision, and selected horizon/execution variants. Emphasize that best iMF-AttnRes reaches 526.22 average reward and 44/100 success-like episodes versus native diffusion policy at 319.2 and 29/100, while several ablations underperform.
\end{tcolorbox}
\caption{Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering.}
\label{fig:sim_transfer_ablation}
\end{figure}
\subsection{Derived speed-quality comparisons}
\label{sec:derived_comparisons}
\Cref{tab:derived_comparisons} lists the derived comparisons used to summarize the tradeoff. On socket insertion, iMF-AttnRes infer3 improves average reward by 1.76$\times$ over the ph16 diffusion-style baseline and by 1.48$\times$ over the ph32 diffusion-style baseline. The same infer3 run improves inference FPS by 22.97$\times$ over the ph16 diffusion-style baseline. The infer2 run is 23.48$\times$ lower latency than the ph16 diffusion-style baseline and 28.45$\times$ higher inference FPS than the ph32 diffusion-style baseline. On object transfer, best iMF-AttnRes improves average reward by 1.65$\times$ and success-like episodes by +15 over native Diffusion Policy.
\begin{table}[t]
\centering
\small
\caption{Derived comparisons from the 100-rollout logs.}
\label{tab:derived_comparisons}
\begin{tabular}{llr}
\toprule
Comparison & Metric & Result \\
\midrule
Socket iMF infer3 vs ph16 diffusion & Avg. reward & 1.76$\times$ \\
Socket iMF infer3 vs ph32 diffusion & Avg. reward & 1.48$\times$ \\
Socket iMF infer3 vs ACT & Avg. reward & 5.23$\times$ \\
Socket iMF infer3 vs SmolVLA & Avg. reward & 3.25$\times$ \\
Socket iMF infer2 vs ph16 diffusion & Latency reduction & 23.48$\times$ \\
Socket iMF infer3 vs ph16 diffusion & Inference FPS & 22.97$\times$ \\
Socket iMF infer2 vs ph32 diffusion & Inference FPS & 28.45$\times$ \\
Sim transfer iMF vs native diffusion & Avg. reward & 1.65$\times$ \\
Sim transfer iMF vs native diffusion & Success-like episodes & +15 \\
ResNet18 iMF vs layer18 baseline & Inference FPS & 237.65$\times$ \\
\bottomrule
\end{tabular}
\end{table}
\section{Limitations}
\label{sec:limitations}
All experiments are simulation rollouts. The data do not establish real-robot transfer, robustness to sensing changes, or safety under hardware execution. CoRL submissions should provide convincing robotics evidence; the present draft should therefore be viewed as a simulation-first manuscript that still needs real-robot or stronger transfer evidence before formal submission.
The speed measurements are also not perfectly hardware-normalized. Runs were collected on RTX 5880 Ada, L20, and RTX 5090 machines as encoded by their run names. Large latency differences between infer100 diffusion-style policies and infer1--3 iMF-AttnRes policies are meaningful because they reflect algorithmic sampling cost, but exact FPS ratios may include hardware and implementation effects. Future experiments should rerun all main policies on the same GPU, with identical rollout parallelism and identical observation preprocessing.
Finally, iMF-AttnRes is sensitive to design choices. In sim\_transfer, several iMF variants underperform the native diffusion-policy baseline, and full AttnRes in the vision encoder performs worse than more conservative policy-transformer AttnRes. This suggests that average-flow training is not a plug-in guarantee; horizon, execution length, visual tokenization, and residual placement must be tuned carefully.
\section{Conclusion}
\label{sec:conclusion}
We presented a first simulation study of iMF-AttnRes for fast robot action generation. The method adapts improved Mean Flow to VLA imitation learning and uses Attention Residuals to replace fixed residual accumulation in the policy transformer. In socket peg insertion, iMF-AttnRes achieves higher average and median reward than diffusion-style infer100 baselines while reducing inference latency from hundreds of milliseconds to roughly 15 ms. In object transfer, the best iMF-AttnRes run improves average reward and success-like episode count over the native diffusion-policy baseline. At the same time, ablations show that the method is sensitive to horizon/execution settings and residual placement. The next step is controlled, hardware-normalized evaluation with real-robot transfer.
\clearpage
\bibliography{refs}
\end{document}
+140
View File
@@ -0,0 +1,140 @@
@article{chi2023diffusion,
title = {Diffusion Policy: Visuomotor Policy Learning via Action Diffusion},
author = {Chi, Cheng and Xu, Zhenjia and Feng, Siyuan and Cousineau, Eric and Du, Yilun and Burchfiel, Benjamin and Tedrake, Russ and Song, Shuran},
year = {2023},
journal = {arXiv preprint arXiv:2303.04137},
eprint = {2303.04137},
archivePrefix = {arXiv}
}
@inproceedings{ho2020denoising,
title = {Denoising Diffusion Probabilistic Models},
author = {Ho, Jonathan and Jain, Ajay and Abbeel, Pieter},
year = {2020},
booktitle = {Advances in Neural Information Processing Systems}
}
@inproceedings{peebles2023scalable,
title = {Scalable Diffusion Models with Transformers},
author = {Peebles, William and Xie, Saining},
year = {2023},
booktitle = {IEEE/CVF International Conference on Computer Vision}
}
@inproceedings{lipman2023flow,
title = {Flow Matching for Generative Modeling},
author = {Lipman, Yaron and Chen, Ricky T. Q. and Ben-Hamu, Heli and Nickel, Maximilian and Le, Matt},
year = {2023},
booktitle = {International Conference on Learning Representations}
}
@article{liu2022flow,
title = {Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow},
author = {Liu, Xingchao and Gong, Chengyue and Liu, Qiang},
year = {2022},
journal = {arXiv preprint arXiv:2209.03003},
eprint = {2209.03003},
archivePrefix = {arXiv}
}
@article{geng2025mean,
title = {Mean Flows for One-step Generative Modeling},
author = {Geng, Zhengyang and Deng, Mingyang and Bai, Xingjian and Kolter, J. Zico and He, Kaiming},
year = {2025},
journal = {arXiv preprint arXiv:2505.13447},
eprint = {2505.13447},
archivePrefix = {arXiv}
}
@article{geng2025improved,
title = {Improved Mean Flows: On the Challenges of Fastforward Generative Models},
author = {Geng, Zhengyang and Lu, Yiyang and Wu, Zongze and Shechtman, Eli and Kolter, J. Zico and He, Kaiming},
year = {2025},
journal = {arXiv preprint arXiv:2512.02012},
eprint = {2512.02012},
archivePrefix = {arXiv}
}
@article{brohan2022rt,
title = {RT-1: Robotics Transformer for Real-World Control at Scale},
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Dabis, Joseph and Finn, Chelsea and Gopalakrishnan, Keerthana and Hausman, Karol and Herzog, Alexander and Hsu, Jasmine and others},
year = {2022},
journal = {arXiv preprint arXiv:2212.06817},
eprint = {2212.06817},
archivePrefix = {arXiv}
}
@inproceedings{brohan2023rt,
title = {RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control},
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Chen, Xi and Choromanski, Krzysztof and Ding, Tianli and Driess, Danny and Dubey, Avinava and Finn, Chelsea and others},
year = {2023},
booktitle = {Conference on Robot Learning}
}
@article{kim2024openvla,
title = {OpenVLA: An Open-Source Vision-Language-Action Model},
author = {Kim, Moo Jin and Pertsch, Karl and Karamcheti, Siddharth and Xiao, Ted and Balakrishna, Ashwin and Nair, Suraj and Rafailov, Rafael and Foster, Ethan and Lam, Grace and Sanketi, Pannag and others},
year = {2024},
journal = {arXiv preprint arXiv:2406.09246},
eprint = {2406.09246},
archivePrefix = {arXiv}
}
@article{shukor2025smolvla,
title = {SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},
author = {Shukor, Mustafa and Aubakirova, Dana and Capuano, Francesco and Kooijmans, Pepijn and Palma, Steven and Zouitine, Adil and Aractingi, Michel and Pascal, Caroline and Russi, Martino and Marafioti, Andres and others},
year = {2025},
journal = {arXiv preprint arXiv:2506.01844},
eprint = {2506.01844},
archivePrefix = {arXiv}
}
@article{black2024pi0,
title = {{$\pi_0$}: A Vision-Language-Action Flow Model for General Robot Control},
author = {Black, Kevin and Brown, Noah and Driess, Danny and Esmail, Adnan and Equi, Michael and Finn, Chelsea and Fusai, Niccolo and Groom, Lachy and Hausman, Karol and Ichter, Brian and others},
year = {2024},
journal = {arXiv preprint arXiv:2410.24164},
eprint = {2410.24164},
archivePrefix = {arXiv}
}
@article{pertsch2025fast,
title = {FAST: Efficient Action Tokenization for Vision-Language-Action Models},
author = {Pertsch, Karl and Stachowicz, Kyle and Ichter, Brian and Driess, Danny and Nair, Suraj and Vuong, Quan and Mees, Oier and Finn, Chelsea and Levine, Sergey},
year = {2025},
journal = {arXiv preprint arXiv:2501.09747},
eprint = {2501.09747},
archivePrefix = {arXiv}
}
@article{zhao2023learning,
title = {Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware},
author = {Zhao, Tony Z. and Kumar, Vikash and Levine, Sergey and Finn, Chelsea},
year = {2023},
journal = {arXiv preprint arXiv:2304.13705},
eprint = {2304.13705},
archivePrefix = {arXiv}
}
@inproceedings{he2016deep,
title = {Deep Residual Learning for Image Recognition},
author = {He, Kaiming and Zhang, Xiangyu and Ren, Shaoqing and Sun, Jian},
year = {2016},
booktitle = {IEEE Conference on Computer Vision and Pattern Recognition}
}
@inproceedings{vaswani2017attention,
title = {Attention Is All You Need},
author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N. and Kaiser, Lukasz and Polosukhin, Illia},
year = {2017},
booktitle = {Advances in Neural Information Processing Systems}
}
@article{kimi2026attention,
title = {Attention Residuals},
author = {{Kimi Team}},
year = {2026},
journal = {arXiv preprint arXiv:2603.15031},
eprint = {2603.15031},
archivePrefix = {arXiv}
}
+19
View File
@@ -0,0 +1,19 @@
# Inputs
The paper-orchestra pipeline expects the four files below in this
directory before you start. Optional figures go in `figures/`.
## Required
- `idea.md` — Idea Summary (Sparse or Dense; see io-contract.md)
- `experimental_log.md` — Setup, raw numeric data, qualitative observations
- `template.tex` — LaTeX template for the target conference
- `conference_guidelines.md` — Page limit, mandatory sections, formatting rules
## Optional
- `figures/` — Pre-existing figures (PNG/PDF). If empty, the
plotting agent generates everything from scratch.
See `skills/paper-orchestra/references/io-contract.md` in the repo for the
exact schemas of each file.
+151
View File
@@ -0,0 +1,151 @@
The Conference on Robot Learning (CoRL) is an annual international conference aiming to bring together the robotics and machine learning research communities. It focuses on the increasingly important role of learning in robotics and its interaction with other areas of robotics. The aim of CoRL 2026 is to publish significant original research at the intersection of robotics and machine learning. CoRL is a selective, single-track international conference addressing theory and practice of machine learning for robots. CoRL welcomes papers in areas such as:
Learning representations for robotic perception and control
Learning robot foundation models or general-purpose knowledge systems for robotics
Imitation learning for robotics, e.g. by behavioral cloning and/or inverse reinforcement learning
Reinforcement learning for control of physical robots
Model-based and model-free learning for robotic control and decision-making
Combination of learning- and planning-based approaches in robotics
Probabilistic learning and representation of uncertainty in robotics
Automatic robotic data generation for learning methods in robotics
Learning for Robot Task and Motion Planning
Learning for multimodal robot perception, sensor fusion, and robot vision
Learning for human-robot interaction and robot instruction by natural language, gestures as well as alternative devices
Learning for hardware design and optimization
Learning approaches to robot safety and alignment; and safety approaches applicable to learning-based robotic systems
Applications of robot learning in robot manipulation, navigation, locomotion, driving, flight, and other areas of robotics
Robot systems, hardware, and sensors for learning and data-driven approaches
Theoretical foundations of robot learning, including generalization theory and uncertainty quantification
Video models and latent world models for robot learning
Benchmarks and datasets for robot learning
Submissions should focus on a core robotics problem and demonstrate the relevance of proposed models, algorithms, datasets, and benchmarks to robotics. Authors are encouraged to report real-robot experiments or provide convincing evidence that simulation experiments are transferable to real robots. Submissions without a robotics focus will be returned without review.
All submissions should include a Limitations section, explicitly describing limiting assumptions, failure modes, and other limitations of the results and experiments and how these might be addressed in the future.
Authors are also encouraged to submit video, code and data as supplementary materials.
For paper submission format and timeline, see: Instruction for Authors
New This Year (details see below)
Paper Submission Requirements and Instructions
Submission Policy
Reciprocal Reviewing Requirement
Rebuttal Process
Concurrent/contemporaneous Work
Camera Ready Paper Submission
Key Dates
New This Year (details see below)
An additional 9th page for camera-ready submissions to incorporate reviewer feedback (initial submission is 8 pages).
Paper Submission Requirements and Instructions
CoRL is double-blind, which means all papers must be anonymized. The accepted papers and reviews will be publicly accessible and de-anonymized after decisions are announced. Our aim is to have at least two reviewers per paper. Accepted papers will appear in the Proceedings of Machine Learning Research (formerly JMLR Workshop and Conference Proceedings).
Please keep in mind that the deadlines are final, and we cannot make any accommodations for missing the abstract deadline or paper deadline. Authors may update the title and abstract on OpenReview after the abstract submission deadline. However, authors cannot be added or removed after this deadline, as this information is required to facilitate reviewer assignments. If you need to request an exception, please email our OpenReview Chair Carlo Sferrazza (csferrazza@austin.utexas.edu).
Papers may be submitted through OpenReview via the link here (https://openreview.net/group?id=robot-learning.org/CoRL/2026/Conference). All submissions should comply with the format and length indicated below:
Page limit will be 8 pages for the main paper. Acknowledgments, References, and Appendix (optional) will not count towards the page limit. Please note that reviewers are not required to read the Appendix. The Appendix can be included with the main paper or with the Supplement.
Please use the LaTeX template: here (https://drive.google.com/file/d/1R9irIW0ImqeDHh5g8Ukm2XGy91sWxOsQ/view?usp=drive_link ).
Authors are encouraged to submit a supplementary file containing further details, which the reviewers may decide to consult.
Authors are highly encouraged to submit a video not exceeding 250 MB (strict) and not more than 3 minutes in length (suggested), providing an overview of the work.
All supplementary materials will be submitted through OpenReview as a single zip file.
All accepted papers will be presented in poster sessions, while selected papers will be invited for an oral spotlight presentation.
All submissions should include a Limitations section (counted toward the 8-page limit), explicitly describing limiting assumptions, failure modes, and other limitations of the results and experiments, and how these might be addressed in the future.
Authors are strongly encouraged to include all relevant details in the main paper and the appendix to ensure that future researchers can reproduce the methodology and results.
Submission Policy
We will not accept papers that are identical or substantially similar to papers that have previously been published or accepted for publication in an archival venue, nor papers submitted in parallel to other conferences or archival venues. Archival venues include conferences and journals with formally published proceedings, but do not include non-archival workshops. Submission is permitted for papers that have previously appeared only as a technical report, e.g., in arXiv.
Dual submission policy: Dual submission is not allowed by default. Submissions that are simultaneously under review at another archival venue will be desk-rejected without review. However, we recognize that some conferences may have decision timelines that slightly overlap with our review period (e.g., ICML decisions on May 1st). In such cases, authors must notify the Publication Chair before submission and clearly indicate the overlapping venue and relevant dates. Approval for such exceptions will be granted on a case-by-case basis. Failure to disclose dual submission may result in rejection and may be reported to both venues.
Generative AI policies: We understand that generative AI is a useful tool in the paper writing process. In order to ensure that the use of generative AI does not pose an undue burden on the review process, we are implementing the following policies this year:
Papers that contain citations to references that do not exist will be desk-rejected after review by the Program Committee.
Authors must take full responsibility for the content in submitted papers. LLMs or other generative AI tools do not qualify for authorship. During the paper submission process, authors will be asked to briefly disclose how generative AI tools were used for research and paper writing.
Authors who submit multiple versions of substantially the same paper (e.g., different papers with the same method, but slightly different ablations, baselines, or algorithmic variations) will face desk rejection of all papers they have submitted, and potentially also be prevented from submitting papers to future iterations of the conference. We are aware that generative AI tools have been used for this purpose in recent conferences, and we will implement checks to flag such papers.
With these policies, we aim to strike a balance between beneficial uses of generative AI (e.g., help with paper editing, figures, literature review, or brainstorming) and uses that degrade the overall quality and integrity of the peer-review process.
Reciprocal Reviewing Requirement
Qualified authors are required to review for CoRL, according to the policies below.
All submissions must include at least one author who agrees to serve as a CoRL reviewer. They are qualified if they have at least one accepted publication at a previous robotics (CoRL/ ICRA/ IROS/ RSS) or learning conference (ICLR/ NeurIPS/ ICML) or equivalent journal. If none of the authors are qualified under this definition or all authors are exempt (e.g., serving as AC or SAC), then they are exempt from this requirement. The abstract submission form will allow submitters to designate an author to fulfill this requirement or to indicate that the submission is exempt from the requirement.
Additionally, every author with 3 or more submissions must agree to serve as a reviewer. Authors are exempt from the review requirement if they serve as an AC, an SAC, or another organizing chair for CoRL 2026.
Submissions that do not meet this reciprocal review requirement may be desk-rejected. Additionally, reviewers who fail to adequately participate in the review process (e.g., not submitting reviews on time or submitting highly insufficient or inappropriate reviews) may have their own submissions desk rejected. The program chairs may grant exceptions on a case-by-case basis.
Rebuttal Process
In order to focus reviewer effort, papers that receive all reviews in the “reject” range in the initial round will not proceed to the rebuttal phase.
Authors of all other papers are invited to submit a 1-page rebuttal in pdf by the rebuttal deadline. You will not be able to give individual responses to the reviewer on OpenReview, nor will you be able to update the paper during the rebuttal process. The rebuttal should be focused on addressing any factual errors in the review.
All rebuttals will be reviewed by the reviewer, the original AC, and, if applicable, external reviewers. The AC reserves the right to invite new reviewers if needed. The results of the current review(s) will be shared with the new reviewers in such cases.
Concurrent/contemporaneous Work
We consider papers contemporaneous if they are published within the last 4 months, so authors do not need to compare their own work to that paper. Authors are encouraged to cite and discuss all relevant papers, but they may be excused for not knowing about papers not published in peer-reviewed conference proceedings or journals, which include papers exclusively available on arXiv.
Camera Ready Paper Submission
New this year — additional page for incorporating reviewer feedback
Congratulations again on getting your paper accepted to CoRL 2026! The camera-ready version of your paper is due by Oct 12, 2026 (23:59 Anywhere on Earth). Please submit your camera-ready PDF and signed permission form through OpenReview.
Here are the instructions for preparing and submitting your camera-ready paper:
1. The camera-ready paper should follow the template: 8-page main text + acknowledgment + references + appendix (optional). Note that this year, we are allowing an additional page to accommodate feedback from the review process. The appendix should be included at the end of the camera-ready PDF, rather than as a separate file. The template is available here. Please make sure to use “\usepackage[final]{corl_2026}.
2. One important note: PMLR does not allow videos to be submitted as supplementary material. If you have videos, code, datasets, and other supplementary materials, please host them on your own (e.g., YouTube, GitHub, etc). You should provide, in the main text of your paper, a link to this material or a link to your project website.
3. Please sign the permission form for publication in PMLR (form available here). Please rename the pdf to <paperid_firstname_lastname>.pdf.
Please ensure:
1. Paper length: 9 pages main text + Acknowledgement + References + Appendix (optional)
2. The author list is NOT anonymous.
3. Footer on the 1st page: "10th Conference on Robot Learning (CoRL 2026), Austin, Texas, USA."
4. Text density, fonts, and spacing should be the same as the provided template on the website.
5. The margin should be the same as the provided template on the website.
6. The OpenReview title is the same as the PDF title.
Please contact Yoonchang Sung (yoonchang.sung@ntu.edu.sg) if you have any questions.
+472
View File
@@ -0,0 +1,472 @@
% File: corl_2026.sty
%
% Latex templates for the Conference on Robot Learning (CoRL)
%
% This template is heavily inspired by the NeurIPS, ICML, ICLR and IEEE Transactions latex templates.
% Hence we would like to thank: Roman Garnett and the previous mantainers of the NIPS style, Percy Liang and the previous mantainers of the ICML style, Hugo Larochelle for the ICLR style, and Michael Shell for the IEEE Transactions style.
%
% History:
% 2017/04/16 - First revision by Roberto Calandra (roberto.calandra@berkeley.edu).
% Main changes:
% - The abstract is more compact compared to NeurIPS/ICML
% - References are by default using natbib with squared numbers (e.g., [1])
% - DOI fields from the bibtex are automatically converted to hyperlinks
% to the corresponding page
% - acknowledgments are now a command, and the corresponding subsubsection is
% automatically included only in the final version
% 2017/06/12 - Modified to use corlabbrvnat.bst, which order the reference by order of appearance in the paper
% 2017/06/13 - fixed typo
% 2018/05/09 - Slightly modified for CoRL 2018 by Jun Morimoto (xmorimo@atr.jp)
% 2019/01/28 - Slightly modified for CoRL 2019 by Jun Nakanishi (jnakanis@meijo-u.ac.jp)
% 2020/02/02 - Slightly modified for CoRL 2020 by Cynthia Matuszek (cmat@umbc.edu)
% 2020/08/19 - Added preprint option by Roberto Calandra (rcalandra@fb.com)
% 2021/05/06 - Slightly modified for CoRL 2021 by Gerhard Neumann (gerhard.neumann@kit.edu)
% 2022/03/09 - Slightly modified for CoRL 2022 by Minas Liarokapis (minas.liarokapis@auckland.ac.nz)
% 2022/03/06 - Slightly modified for CoRL 2023 by Marc Toussaint (toussaint@tu-berlin.de)
% 2024/03/26 - Slightly modified for CoRL 2024 by David Held (dheld@andrew.cmu.edu)
% 2026/01/10 - Slightly modified for CoRL 2026 by Yoonchang Sung (yoonchang.sung@ntu.edu.sg)
%
% TODO: nohyperref is not working at the moment
%
\NeedsTeXFormat{LaTeX2e}
% Content to be changed from year to year
\ProvidesPackage{corl_2026}[2026/08/15 CORL2026 submission/preprint/camera-ready style file]
\newcommand{\@conferenceordinal}{10th}
\newcommand{\@conferenceyear}{2026}
\newcommand{\@conferencelocation}{Austin TX, USA}
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Accepted options: [final,preprint,nonatbib,nohyperref]
% Declare the final option, which creates camera-ready copy
\newif\if@conferencefinal\@conferencefinalfalse
\DeclareOption{final}{
\@conferencefinaltrue
}
% Declare the preprint option, which creates a camera-ready copy without the corl footnote
\newif\if@preprinttype\@preprinttypefalse
\DeclareOption{preprint}{
\@preprinttypetrue
}
% The natbib package is loaded by default. Declaring the nonatbib option, does not load natbib in case of package clash (users can pass options to natbib via \PassOptionsToPackage)
\newif\if@natbib\@natbibtrue
\DeclareOption{nonatbib}{
\@natbibfalse
}
% The hyperref package is loaded by default. Declaring the nohyperref option, does not load the hyperref.
\DeclareOption{nohyperref}{%
\gdef\nohyperref{1}
}
% Activate the options
\ProcessOptions\relax
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Required packages:
\RequirePackage{lineno}
\RequirePackage{color}
% Load natbib unless told otherwise
\if@natbib
\RequirePackage[square,numbers]{natbib}
\bibliographystyle{corlabbrvnat}
\fi
% set page geometry
\RequirePackage{hyperref} % hyperlinks
\RequirePackage[verbose=true,letterpaper]{geometry}
\AtBeginDocument{
\newgeometry{
textheight=9in,
textwidth=5.5in,
top=1in,
headheight=12pt,
headsep=25pt,
footskip=30pt
}
\@ifpackageloaded{fullpage}
{\PackageWarning{corl_2026}{fullpage package not allowed! Overwriting formatting.}}
{}
}
\ifdefined\nohyperref\else\ifdefined\hypersetup
\definecolor{mydarkblue}{rgb}{0,0.08,0.45}
\hypersetup{ %
pdftitle={},
pdfauthor={},
pdfsubject={Proceedings of the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear)},
pdfkeywords={},
pdfborder=0 0 0,
pdfpagemode=UseNone,
colorlinks=true,
linkcolor=mydarkblue,
citecolor=mydarkblue,
filecolor=mydarkblue,
urlcolor=mydarkblue,
pdfview=FitH}
\ifdefined\isaccepted \else
\hypersetup{pdfauthor={Anonymous Submission}}
\fi
\fi\fi
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%
% fonts
\renewcommand{\rmdefault}{ptm}
\renewcommand{\sfdefault}{phv}
% Create acknowledgments -- only if the option 'final' is activated
\providecommand{\acknowledgments}{}
\renewcommand{\acknowledgments}[1]{%
\if@conferencefinal%
\subsubsection*{Acknowledgments} #1
\fi
\if@preprinttype%
\subsubsection*{Acknowledgments} #1
\fi
}
% handle tweaks for camera-ready copy vs. submission copy
\if@conferencefinal
\newcommand{\@noticestring}{%
\@conferenceordinal\/ Conference on Robot Learning
(CoRL \@conferenceyear), \@conferencelocation.%
}
\else
\if@preprinttype
\newcommand{\@noticestring}{%
% Nothing here.
}
\else
\newcommand{\@noticestring}{%
Submitted to the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear). Do not distribute.%
}
% line numbers for submission
\linenumbers
% fix incompatibilities between lineno and amsmath, if required, by
% transparently wrapping linenomath environments around amsmath
% environments
\AtBeginDocument{%
\@ifpackageloaded{amsmath}{%
\newcommand*\patchAmsMathEnvironmentForLineno[1]{%
\expandafter\let\csname old#1\expandafter\endcsname\csname #1\endcsname
\expandafter\let\csname oldend#1\expandafter\endcsname\csname end#1\endcsname
\renewenvironment{#1}%
{\linenomath\csname old#1\endcsname}%
{\csname oldend#1\endcsname\endlinenomath}%
}%
\newcommand*\patchBothAmsMathEnvironmentsForLineno[1]{%
\patchAmsMathEnvironmentForLineno{#1}%
\patchAmsMathEnvironmentForLineno{#1*}%
}%
\patchBothAmsMathEnvironmentsForLineno{equation}%
\patchBothAmsMathEnvironmentsForLineno{align}%
\patchBothAmsMathEnvironmentsForLineno{flalign}%
\patchBothAmsMathEnvironmentsForLineno{alignat}%
\patchBothAmsMathEnvironmentsForLineno{gather}%
\patchBothAmsMathEnvironmentsForLineno{multline}%
}{}
}
\fi
\fi
% The DOI will now automatically generate a URL link. Pretty cool!
% + Fix for hyperref and DOI (https://www.tug.org/pipermail/tex-live/2012-August/032161.html)
%-----------------------------
\makeatletter
\providecommand{\doi}[1]{%
\begingroup
\let\bibinfo\@secondoftwo
\urlstyle{rm}%
\href{http://dx.doi.org/#1}{%
doi:\discretionary{}{}{}%
\nolinkurl{#1}%
}%
\endgroup
}
% \makeatother
% %-----------------------------
\widowpenalty=10000
\clubpenalty=10000
\flushbottom
\sloppy
% font sizes with reduced leading
\renewcommand{\normalsize}{%
\@setfontsize\normalsize\@xpt\@xipt
\abovedisplayskip 7\p@ \@plus 2\p@ \@minus 5\p@
\abovedisplayshortskip \z@ \@plus 3\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 4\p@ \@plus 3\p@ \@minus 3\p@
}
\normalsize
\renewcommand{\small}{%
\@setfontsize\small\@ixpt\@xpt
\abovedisplayskip 6\p@ \@plus 1.5\p@ \@minus 4\p@
\abovedisplayshortskip \z@ \@plus 2\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 3\p@ \@plus 2\p@ \@minus 2\p@
}
\renewcommand{\footnotesize}{\@setfontsize\footnotesize\@ixpt\@xpt}
\renewcommand{\scriptsize}{\@setfontsize\scriptsize\@viipt\@viiipt}
\renewcommand{\tiny}{\@setfontsize\tiny\@vipt\@viipt}
\renewcommand{\large}{\@setfontsize\large\@xiipt{14}}
\renewcommand{\Large}{\@setfontsize\Large\@xivpt{16}}
\renewcommand{\LARGE}{\@setfontsize\LARGE\@xviipt{20}}
\renewcommand{\huge}{\@setfontsize\huge\@xxpt{23}}
\renewcommand{\Huge}{\@setfontsize\Huge\@xxvpt{28}}
% sections with less space
\providecommand{\section}{}
\renewcommand{\section}{%
\@startsection{section}{1}{\z@}%
{-2.0ex \@plus -0.5ex \@minus -0.2ex}%
{ 1.5ex \@plus 0.3ex \@minus 0.2ex}%
{\large\bf\raggedright}%
}
\providecommand{\subsection}{}
\renewcommand{\subsection}{%
\@startsection{subsection}{2}{\z@}%
{-1.8ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.8ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\subsubsection}{}
\renewcommand{\subsubsection}{%
\@startsection{subsubsection}{3}{\z@}%
{-1.5ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.5ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\paragraph}{}
\renewcommand{\paragraph}{%
\@startsection{paragraph}{4}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subparagraph}{}
\renewcommand{\subparagraph}{%
\@startsection{subparagraph}{5}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subsubsubsection}{}
\renewcommand{\subsubsubsection}{%
\vskip5pt{\noindent\normalsize\rm\raggedright}%
}
% float placement
\renewcommand{\topfraction }{0.85}
\renewcommand{\bottomfraction }{0.4}
\renewcommand{\textfraction }{0.1}
\renewcommand{\floatpagefraction}{0.7}
\newlength{\@nipsabovecaptionskip}\setlength{\@nipsabovecaptionskip}{7\p@}
\newlength{\@nipsbelowcaptionskip}\setlength{\@nipsbelowcaptionskip}{\z@}
\setlength{\abovecaptionskip}{\@nipsabovecaptionskip}
\setlength{\belowcaptionskip}{\@nipsbelowcaptionskip}
% swap above/belowcaptionskip lengths for tables
\renewenvironment{table}
{\setlength{\abovecaptionskip}{\@nipsbelowcaptionskip}%
\setlength{\belowcaptionskip}{\@nipsabovecaptionskip}%
\@float{table}}
{\end@float}
% footnote formatting
\setlength{\footnotesep }{6.65\p@}
\setlength{\skip\footins}{9\p@ \@plus 4\p@ \@minus 2\p@}
\renewcommand{\footnoterule}{\kern-3\p@ \hrule width 12pc \kern 2.6\p@}
\setcounter{footnote}{0}
% paragraph formatting
\setlength{\parindent}{\z@}
\setlength{\parskip }{5.5\p@}
% list formatting
\setlength{\topsep }{4\p@ \@plus 1\p@ \@minus 2\p@}
\setlength{\partopsep }{1\p@ \@plus 0.5\p@ \@minus 0.5\p@}
\setlength{\itemsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\parsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\leftmargin }{3pc}
\setlength{\leftmargini }{\leftmargin}
\setlength{\leftmarginii }{2em}
\setlength{\leftmarginiii}{1.5em}
\setlength{\leftmarginiv }{1.0em}
\setlength{\leftmarginv }{0.5em}
\def\@listi {\leftmargin\leftmargini}
\def\@listii {\leftmargin\leftmarginii
\labelwidth\leftmarginii
\advance\labelwidth-\labelsep
\topsep 2\p@ \@plus 1\p@ \@minus 0.5\p@
\parsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\itemsep \parsep}
\def\@listiii{\leftmargin\leftmarginiii
\labelwidth\leftmarginiii
\advance\labelwidth-\labelsep
\topsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\parsep \z@
\partopsep 0.5\p@ \@plus 0\p@ \@minus 0.5\p@
\itemsep \topsep}
\def\@listiv {\leftmargin\leftmarginiv
\labelwidth\leftmarginiv
\advance\labelwidth-\labelsep}
\def\@listv {\leftmargin\leftmarginv
\labelwidth\leftmarginv
\advance\labelwidth-\labelsep}
\def\@listvi {\leftmargin\leftmarginvi
\labelwidth\leftmarginvi
\advance\labelwidth-\labelsep}
% create title
\providecommand{\maketitle}{}
\renewcommand{\maketitle}{%
\par
\begingroup
\renewcommand{\thefootnote}{\fnsymbol{footnote}}
% for perfect author name centering
\renewcommand{\@makefnmark}{\hbox to \z@{$^{\@thefnmark}$\hss}}
% The footnote-mark was overlapping the footnote-text,
% added the following to fix this problem (MK)
\long\def\@makefntext##1{%
\parindent 1em\noindent
\hbox to 1.8em{\hss $\m@th ^{\@thefnmark}$}##1
}
\thispagestyle{empty}
\@maketitle
\@thanks
\@notice
\endgroup
\let\maketitle\relax
\let\thanks\relax
}
% rules for title box at top of first page
\newcommand{\@toptitlebar}{
\hrule height 4\p@
\vskip 0.25in
\vskip -\parskip%
}
\newcommand{\@bottomtitlebar}{
\vskip 0.29in
\vskip -\parskip
\hrule height 1\p@
\vskip 0.09in%
}
%% keywords as first class citizens
\def\keywords#1{%
% \ifdefined\isaccepted \else
% \par {\bf Keywords:} #1%
% \fi
% \ifdefined\nohyperref\else\ifdefined\hypersetup
% \hypersetup{pdfkeywords={#1}}
% \fi\fi
\ifdefined\isaccepted \else
\begin{quote}
\textbf{Keywords:} #1%
\end{quote}
\fi
\ifdefined\nohyperref\else\ifdefined\hypersetup
\hypersetup{pdfkeywords={#1}}
\fi\fi
}
% create title (includes both anonymized and non-anonymized versions)
\providecommand{\@maketitle}{}
\renewcommand{\@maketitle}{%
\vbox{%
\hsize\textwidth
\linewidth\hsize
\vskip 0.1in
% \@toptitlebar
\centering
{\LARGE\bf \@title\par}
% \@bottomtitlebar
\if@conferencefinal
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\if@preprinttype
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}
Anonymous Author(s) \\
Affiliation \\
Address \\
\texttt{email} \\
\end{tabular}%
\fi
\fi
\vskip 0.3in \@minus 0.1in
}
}
% add conference notice to bottom of first page
\newcommand{\ftype@noticebox}{8}
\newcommand{\@notice}{%
% give a bit of extra room back to authors on first page
\enlargethispage{2\baselineskip}%
\@float{noticebox}[b]%
\footnotesize\@noticestring%
\end@float%
}
% abstract styling
\renewenvironment{abstract}%
{%
% \vskip 0.075in%
% \centerline%
% {\large\bf Abstract}%
% \vspace{0.5ex}%
\begin{quote}%
\textbf{Abstract:}%
}
{
\par%
\vskip 1ex%
% \ifdefined\keywords
% \textbf{Keywords:}%
% \@keywords%
% \else
% \fi
\end{quote}%
}
\endinput
Binary file not shown.
File diff suppressed because it is too large Load Diff
+20
View File
@@ -0,0 +1,20 @@
% This file was created with JabRef 2.10.
% Encoding: UTF-8
@Article{Gauss1857,
Title = {Theory of the motion of the heavenly bodies moving about the sun in conic sections},
Author = {Carl Friedrich Gauss and Charles Henry Davis},
Journal = {Gauss's Theoria Motus},
Year = {1857},
Number = {1},
Pages = {5--23},
Volume = {76}
}
@Book{Lagrange1788,
title = {M{\'e}canique Analytique},
author = {Joseph-Louis Lagrange},
publisher = {Desaint, Paris},
year = {1788}
}
+92
View File
@@ -0,0 +1,92 @@
# Experimental Log: iMF-AttnRes for Fast Vision-Language-Action Imitation Learning in RoboIMI
## 1. Experimental Setup
We evaluate imitation-learning policies in the RoboIMI simulated manipulation benchmark. The study focuses on two environments:
- **socket_peg / socket-insert**: a socket peg insertion task.
- **sim_transfer / object transfer**: a block/object transfer manipulation task.
The central method is an iMF-AttnRes VLA policy that combines improved Mean Flow (iMF) training with Attention Residuals (AttnRes). The comparison set includes diffusion-policy-style DiT baselines, ACT, SmolVLA, and several ablations of the iMF-AttnRes architecture and horizon/execution settings. Each reported result is based on **100 simulation rollouts** unless otherwise stated.
Reported metrics:
- `avg_reward`: mean cumulative episode reward over 100 rollouts.
- `median_reward`: median cumulative episode reward.
- `avg_max_reward`: mean maximum per-episode reward statistic reported by the rollout evaluator.
- `max_reward`: highest cumulative episode reward among 100 rollouts.
- `nonzero_reward`: number of episodes with nonzero reward.
- `success_like`: number of episodes whose max reward crosses the environment threshold (`max_reward > 4` for socket-insert; `max_reward >= 4` for sim_transfer, as recorded in the logs).
- `avg_inference_fps`, `avg_control_fps`, and `avg_inference_time_ms`: rollout speed measurements. Some sim_transfer logs report FPS but not inference time in milliseconds.
Hardware varies across runs and is encoded in run names: 5880g0/5880g1 denotes RTX 5880 Ada runs, l20g* denotes L20 runs, and 5090 denotes RTX 5090 runs. Because hardware differs across experiments, speed comparisons should be interpreted as practical rollout measurements rather than perfectly hardware-normalized measurements.
## 2. Raw Numeric Data
### Table 1: Socket peg insertion, 100-rollout performance and speed
| run_name | method_family | horizon_or_steps | avg_reward | median_reward | avg_max_reward | max_reward | nonzero_reward | success_like | avg_inference_fps | avg_control_fps | avg_inference_time_ms |
|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
| socket-insert-imf-attnres-ph32-exec16-emb384-l12-infer1-unfreeze-step50k-roll1x10-5880g01-20260506-195806 | iMF-AttnRes | infer1, ph32, exec16, 50k | 1275.22 | 1490.5 | 3.12 | 2250.0 | 84/100 | 0/100 | 64.226 | 5.863 | 16.186 |
| socket-insert-no-pretrain-ph16-emb384-l18-infer100-unfreeze-step150k-roll1x10-5090-20260506-193753 | Diffusion-style VLA | infer100, ph16, 150k | 861.53 | 668.0 | 2.58 | 2241.0 | 97/100 | 2/100 | 2.956 | 2.591 | 338.291 |
| socket-insert-no-pretrain-ph32-exec16-emb384-l18-infer100-unfreeze-step150k-roll1x10-5880g1-20260507-092227 | Diffusion-style VLA | infer100, ph32, exec16, 150k | 1019.39 | 804.0 | 2.47 | 2385.0 | 92/100 | 4/100 | 2.516 | 1.928 | 397.912 |
| socket-insert-imf-attnres-ph32-exec16-emb384-l12-infer2-drop005-unfreeze-step150k-roll1x10-5880g01-20260508-165022 | iMF-AttnRes | infer2, ph32, exec16, 150k | 1472.62 | 1825.0 | 3.29 | 2365.0 | 83/100 | 5/100 | 71.579 | 7.181 | 14.409 |
| socket-insert-imf-attnres-ph32-exec16-emb384-l12-infer3-drop005-unfreeze-step150k-roll1x10-5880g01-20260509-172423 | iMF-AttnRes | infer3, ph32, exec16, 150k | 1513.56 | 1901.5 | 3.28 | 2285.0 | 83/100 | 3/100 | 67.891 | 7.247 | 15.122 |
| act-socket-peg-224-20260508-170237 | ACT | action chunking | 289.6 | 13.0 | 1.29 | 2091.0 | 55/100 | 1/100 | 10.5171 | 2.2782 | 100.2116 |
| smolvla_socket_peg_bs80_100k_20260511_145727 | SmolVLA | 100k | 466.16 | 106.0 | 1.72 | 4.0 | 89/100 | 0/100 | 382.53 | 16.84 | 2.614 |
### Table 2: Object transfer / sim_transfer, 100-rollout performance and speed
| run_name | method_family | horizon_or_steps | avg_reward | median_reward | avg_max_reward | max_reward | nonzero_reward | success_like | avg_inference_fps | avg_control_fps | avg_inference_time_ms |
|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
| diffusion_policy_native_dit_ddpm_resnet_best | Diffusion Policy native DiT + DDPM + ResNet | best checkpoint | 319.2 | 6.0 | 1.68 | 1416.0 | 55/100 | 29/100 | 32.09 | 16.70 | n/a |
| embed384_layer18_best_checkpoint | Diffusion-style VLA | emb384, layer18 | 233.52 | 0.0 | 1.12 | 1436.0 | 39/100 | 17/100 | 1.859 | 1.649 | n/a |
| resnet18-multitoken-imf-emb256-l16-ph16-ex08-roll10-l20g2-20260406-112815 | iMF multi-token ResNet18 | step34999, ph16, exec08 | 260.66 | 0.0 | 1.12 | 1422.0 | 33/100 | 23/100 | 441.80 | 13.37 | n/a |
| imf-p2-full-attnres-vision-ph16-ex08-emb384-l12-b40-lr1p25e4-ms50k-l20g3-20260405-002424 | iMF full AttnRes vision | ph16, exec08, 50k | 228.42 | 0.0 | 0.94 | 1482.0 | 31/100 | 16/100 | 55.997 | 4.301 | n/a |
| imf-p1-ph16-ex16-emb384-l12-ms50k-l20g0-20260404-131223 | iMF-AttnRes DiT only | ph16, exec16, 50k | 240.64 | 0.0 | 1.02 | 1428.0 | 32/100 | 19/100 | 137.996 | 4.523 | n/a |
| imf-p1-ph32-ex08-emb384-l12-ms50k-l20g1-20260404-131223 | iMF-AttnRes DiT only | ph32, exec08, 50k | 163.28 | 0.0 | 0.76 | 1566.0 | 26/100 | 12/100 | 85.638 | 4.317 | n/a |
| imf-p1-ph32-ex32-emb384-l12-ms50k-5090-20260404-13122 | iMF-AttnRes DiT only | ph32, exec32, 50k | 260.72 | 0.0 | 1.18 | 1342.0 | 38/100 | 21/100 | 9.557 | 3.299 | n/a |
| imf-p1-ph16-ex08-emb384-l12-ms50k-5880g1-20260404-131223 | iMF-AttnRes DiT only | ph16, exec08, 50k | 229.56 | 0.0 | 1.18 | 1564.0 | 41/100 | 18/100 | 69.984 | 8.053 | n/a |
| imf-p1-ph08-ex08-emb384-l12-ms50k-5880g0-20260404-131223 | iMF-AttnRes DiT only | ph08, exec08, 50k | 237.02 | 0.0 | 1.32 | 1400.0 | 48/100 | 18/100 | 69.192 | 8.173 | n/a |
| imf-p1-ph32-ex16-emb384-l12-ms50k-l20g2-20260404-131223 | iMF-AttnRes DiT only | ph32, exec16, 50k | 49.88 | 0.0 | 0.32 | 1284.0 | 13/100 | 4/100 | 138.903 | 4.482 | n/a |
| sim-transfer-imf-attnres-ph32-exec16-emb384-l12-infer1-unfreeze-step50k-roll5x5-20260403-14130 | iMF-AttnRes | infer1, ph32, exec16, 50k | 526.22 | 86.0 | 2.14 | 1460.0 | 63/100 | 44/100 | 8.911 | 3.193 | n/a |
### Table 3: Derived comparisons used in the paper draft
| comparison | metric | numerator_run | denominator_run | numerator_value | denominator_value | derived_ratio_or_delta |
|---|---|---|---|---:|---:|---:|
| socket best iMF vs ph16 diffusion baseline | avg_reward ratio | iMF infer3 150k | diffusion ph16 infer100 150k | 1513.56 | 861.53 | 1.76x |
| socket best iMF vs ph32 diffusion baseline | avg_reward ratio | iMF infer3 150k | diffusion ph32 infer100 150k | 1513.56 | 1019.39 | 1.48x |
| socket best iMF vs ACT | avg_reward ratio | iMF infer3 150k | ACT | 1513.56 | 289.6 | 5.23x |
| socket best iMF vs SmolVLA | avg_reward ratio | iMF infer3 150k | SmolVLA | 1513.56 | 466.16 | 3.25x |
| socket iMF infer2 vs ph16 diffusion baseline | inference time reduction | iMF infer2 150k | diffusion ph16 infer100 150k | 14.409 | 338.291 | 23.48x lower latency |
| socket iMF infer3 vs ph16 diffusion baseline | inference fps ratio | iMF infer3 150k | diffusion ph16 infer100 150k | 67.891 | 2.956 | 22.97x |
| socket iMF infer2 vs ph32 diffusion baseline | inference fps ratio | iMF infer2 150k | diffusion ph32 infer100 150k | 71.579 | 2.516 | 28.45x |
| sim_transfer best iMF vs native diffusion policy | avg_reward ratio | sim-transfer iMF infer1 | native diffusion policy | 526.22 | 319.2 | 1.65x |
| sim_transfer best iMF vs native diffusion policy | success_like delta | sim-transfer iMF infer1 | native diffusion policy | 44 | 29 | +15 episodes |
| sim_transfer ResNet18 multitoken iMF vs embed384 layer18 baseline | inference fps ratio | ResNet18 multitoken iMF | embed384 layer18 | 441.80 | 1.859 | 237.65x |
## 3. Qualitative Observations
Socket peg insertion shows the clearest quality-speed tradeoff improvement. The iMF-AttnRes variants with ph32/exec16 and only 1--3 inference evaluations achieve substantially higher average and median rewards than the diffusion-style infer100 baselines, while reducing latency from hundreds of milliseconds to roughly 14--16 ms. The infer2 and infer3 variants are the strongest socket policies by average reward, with the infer3 run obtaining avg_reward 1513.56 and median_reward 1901.5. The infer2 run has the fastest iMF-AttnRes socket latency among the high-reward variants at 14.409 ms.
The socket experiments also reveal that nonzero reward alone is not a sufficient quality metric. The diffusion-style ph16 baseline has 97/100 nonzero episodes but much lower median reward than the best iMF-AttnRes runs. Conversely, iMF-AttnRes has 83/100 nonzero episodes but much higher median and average reward, suggesting that when it engages successfully with the task it produces stronger progress toward insertion.
SmolVLA is extremely fast in inference FPS and control FPS, but its reward is lower than iMF-AttnRes in these experiments. ACT is slower than SmolVLA and underperforms iMF-AttnRes in average and median reward.
In sim_transfer, the best recorded iMF-AttnRes run, `sim-transfer-imf-attnres-ph32-exec16-emb384-l12-infer1-unfreeze-step50k-roll5x5-20260403-14130`, reaches avg_reward 526.22 and 44/100 success-like episodes, exceeding the native diffusion-policy DiT/DDPM/ResNet baseline with avg_reward 319.2 and 29/100 success-like episodes. However, several iMF ablations underperform the native diffusion-policy baseline on average reward, indicating that the method is sensitive to horizon/execution choices, vision-token design, and whether AttnRes is applied only in the policy transformer or throughout the vision stack.
For sim_transfer, the ResNet18 multi-token iMF variant is a particularly fast policy at 441.80 inference FPS but does not match the best iMF-AttnRes reward. The full-AttnRes vision variant underperforms the more conservative DiT-only AttnRes settings, suggesting that replacing residual pathways inside the vision encoder may be harmful or require different training hyperparameters.
### Iteration History
1. Early sim_transfer experiments compared diffusion-style baselines against initial iMF-AttnRes policy variants and explored policy horizon/execution length combinations.
2. Subsequent ablations tested multi-token ResNet18 visual encoding, full AttnRes in the vision backbone, and DiT-only AttnRes. These results showed that iMF speedups are robust but reward quality is sensitive to model and horizon choices.
3. Later socket-insert experiments scaled to 100-rollout evaluation and compared iMF-AttnRes against diffusion-style VLA, ACT, and SmolVLA. These experiments provided the strongest evidence for the paper's main claim: iMF-AttnRes can substantially reduce inference latency while improving or preserving manipulation reward.
## 4. Caveats for Paper Writing
- Do not claim real-robot validation; all reported experiments are simulation rollouts.
- Hardware differs across runs, so speed measurements are useful but not fully hardware-normalized.
- Success-like episode counts use thresholds as recorded by each environment's evaluation script: socket-insert logs report `max_reward > 4`, whereas sim_transfer logs report `max_reward >= 4`.
- The reported data are sufficient for a first paper draft but should be supplemented later with controlled hardware-normalized speed measurements and more seeds if the paper is prepared for formal submission.
+73
View File
@@ -0,0 +1,73 @@
本文的主要idea是使用imf和attnres来加速流匹配的推理速度,同时保持相近的推理质量,imf来自于计算机视觉中的he kaiming的研究Improved mean flows: On the challenges of fastforward generative models,之前是用于流匹配图像生成模型的改进,其前述相关研究如下
## MeanFlow [@geng2025meanflow]
Diffusion的生成模型是基于一步一步的方式生成的,这显然不够快,对flow matching中的核心ODE公式
$$\frac{dx_t}{dt} = v(x_t, t)$$
这里提一下在看这篇文章时突然产生的对微分算子$d$的理解,$x_t$表示的是某一时刻的具体位置,我们想知道这一时刻的瞬时速度,要根据速度的计算公式进行计算,速度的计算公式是位移时间比,但$x_t$表示的是某一时刻的具体位置,不是一段时间的位移,因此我们使用微分算子表示在$x_t$附近取很小的一段位移除以很小的一段时间,这里一定注意加入微分算子前后的含义变化,对$t$来说加入微分算子前表示的是具体的时刻$t$,而加入后表示的是一段时间$t$。如果取的足够小,通过这种方式计算出来的一小段内的速度就约等于瞬时速度,这是微分算子的作用。
回到ODE公式本身,可以看出这个公式其实计算的是每一时刻的瞬时速度,而原本的flow matching需要多步采样就来自这个时刻的瞬时性,我们得到瞬时速度后是通过一种近似采样的方式向前移动,是把这个瞬时速度视为一段时间内的平均速度,这也是为什么一开始的模型采样步数多效果就好,因为采样步数太少会导致这种估计非常不准确,使得路径偏移严重。为了解决这种问题,Rectified Flow可译为重整流直接将路径假设为一条直线,注意此处的直线是一种假设,我们只是希望模型能够学会按照这条直线去走,为了能够让模型学会尽可能的按照直线去走,我们在计算损失时让模型预测的速度尽可能接近直线路径的速率,另一方面还通过重整修正,让训练好的但预测路径还不那么直的模型继续学习更直的路线(通过让第一次训练好的模型从任意起点出发得到一个目标点后学习二者间的直线来重整)。
尽管如此,实际实验中发现,模型学习的路线仍然不可能是我们预期中的直线,因此,这种假设可能本身存在一定的问题,真实的分布转移路径可能并不直,那是否有办法让模型可以学习到真实的平均速度而不是基于路径直线这一假设的平均速度呢?
回到平均速度的定义,平均速度是位移除以时间,那么任意两个时间点之间的平均速度是
$$\bar{v}(z_t, r, t) = \frac{1}{t-r} \int_r^t v(z_\tau, \tau)d\tau$$
我们训练模型去拟合这个平均速度,但写出损失函数后可以发现,我们并不知道这个平均速度的真实值(因为没有假设条件了,因此路径可能是任意的)。但MeanFlow提出,可以通过构建平均速度和瞬时速度的关系来获得真实的平均速度。推导如下
$$\begin{aligned} & \bar{v}(z_t, r, t) = \frac{1}{t-r} \int_r^t v(z_\tau, \tau) d\tau \\ \Leftrightarrow \quad & (t - r)\bar{v}(z_t, r, t) = \int_r^t v(z_\tau, \tau) d\tau \\ \Leftrightarrow \quad & \frac{d}{dt} \left[ (t - r)\bar{v}(z_t, r, t) \right] = \frac{d}{dt} \int_r^t v(z_\tau, \tau) d\tau \\ \Leftrightarrow \quad & \bar{v}(z_t, r, t) + (t - r)\frac{d}{dt}\bar{v}(z_t, r, t) = v(z_t, t) \\ \Leftrightarrow \quad & \bar{v}(z_t, r, t) = v(z_t, t) - (t - r)\frac{d}{dt}\bar{v}(z_t, r, t) \end{aligned} \tag{1}$$
其中
$$\frac{d}{dt}\bar{v}(z_t, r, t) = \frac{d\bar{v}}{dz} \cdot \frac{dz}{dt} + \frac{d\bar{v}}{dr} \cdot \frac{dr}{dt} + \frac{d\bar{v}}{dt} \cdot \frac{dt}{dt} = \frac{d\bar{v}}{dz} \cdot v(z_t, t) + \frac{d\bar{v}}{dt}$$
$$= jvp(\bar{v}, (z, r, t), (v, 0, 1))$$
最后可得
$$\bar{v}(z_t, r, t) = v(z_t, t) - (t - r)jvp(\bar{v}, (z, r, t), (v, 0, 1)) \tag{2}$$
这个表达式是自举的,即右式中包含左式,左右相关。式中的瞬时速度$v$我们用条件速度$\epsilon-x_0$代替,结合如下的损失函数
$$L = \mathbb{E}\|\bar{v}_\theta(z_t, r, t) - sg(v(z_t, t) - (t - r)jvp(\bar{v}_\theta, (z, r, t), (v, 0, 1)))\|_2^2$$
我们可以发现,这个损失函数表示的是一种自我纠正,即我们已知起始和终止位置的情况下的条件瞬时速度始终是$v$,模型预测的平均速度是空间中的宏观流,表示在$z$位置从时间$r$到$t$,平均的流向和平均流向的变化率是什么样的,我们用这个特定样本的瞬时速度校正这个平均速度应该是什么样的,和模型预测的平均速度的差值,最终希望的是模型在这一点的平均速度能够逆推出该特定样本的特定瞬时速度。
我们此时会考虑这样一个问题,既然我们想要让模型预测平均速度,为何不直接让模型去拟合直线平均速度$\epsilon-x$,而是费劲进行表达式的推导呢。
原因在于上式中的瞬时速度,从模型最终推理的角度来说,应该是边缘速度场,我们现在考虑的是真实的生成过程中的瞬时速度和平均速度的关系,而边缘速度场不可能是平坦的,因此不能将平均速度场简单的认为和条件速度一致。训练算法如下
此处模型学会的平均速度不再是一个固定值,而是虽然变化但其变化能符合我们特定样本的条件瞬时速度的平均速度。
尽管如此,这部分理解起来仍然怪怪的,并不是十分有说服力。这大概也是后续improved meanflow中提出改进的原因。
可得到训练算法:
![](https://pic1.imgdb.cn/item/69ab8e9959f896a650d454ac.png)
其中和flow matching的主要区别在于加入了雅可比向量积的计算,引入了约16%的额外计算量,但得到的好处是生成时只需一步生成。
## iMF [@gengImprovedMeanFlows2025]
本篇文章是对MeanFlow的改进工作,MeanFlow的推导过程中对平均速度表达式中的瞬时速度都直接使用条件速度替代,实际上原表达式中的所有瞬时速度都应该是边际速度(边缘概率)而不应该是条件速度,在Flow Matching训练时使用条件速度的原因是条件速度的期望和边际速度二者的梯度是相同的,仅相差常数。
但在式2中,jvp内的瞬时速度如果使用条件速度,jvp表达式是一个复杂的非线性表达式,此处的瞬时速度本应该使用边际速度,但这里使用条件速度进行替代,而非线性表达式的特性导致该表达式的期望的梯度并不和使用边际速度时的梯度相同,这就导致训练不稳定。
因此可以对式1进行简单变形,得到瞬时速度和平均速度之间的关系,如下
$$\mathbf{v}(z_t) = \mathbf{u}(z_t) + (t - r)\frac{d}{dt}\mathbf{u}(z_t) $$
$$\mathbf{V}_\theta(z_t) \triangleq \mathbf{u}_\theta(z_t) + (t - r)JVP_{sg}(\mathbf{u}_\theta; \mathbf{v}_\theta) \quad (12)$$
这里$v$和$t$使用同一个网络预测,时刻记住meanflow中预测平均速度的网络$u$的输入是$t,r,z_t$三个参数。当$r=t$时,网络预测的就相当于瞬时速度,因此使用同一个网络预测两遍分别预测$v$和$t$,根据式12得到修正后的瞬时速度,让这个修正后的瞬时速度尽可能接近在直线路径假设下的条件速度$\epsilon-x$。
这里我们已经如果能让网络直接预测$v_\theta$为什么还要通过式12来计算这个瞬时速度呢,这里要结合我们的训练过程考虑,我们的最终目的仍然和MeanFlow一致,即让网络可以预测平均速度,这里我们看似是在计算瞬时速度的损失,实际上该瞬时速度是由式12的右式得来的,而且jvp不参与梯度计算即其中模型预测的瞬时速度$v_\theta$不会产生梯度更新,梯度更新仅发生在前面的$u_\theta(z_t)$中,因此我们实际上是在要求网络学会预测平均速度$u_\theta$。
通过这种改进,我们可以充分利用原本的Flow Matching的训练稳定性。
作者还发现,通过去掉adaLN-zero,但使用将同一个条件token重复多次并作为序列的一部分的形式来进行建模,可以达到和使用adaLN-zero一样的效果。
![](https://pic1.imgdb.cn/item/69c4de9d2fb41b9ea32f5f12.png)
但之前都是用于计算机视觉,我们将其用于机器人的模仿学习和动作生成,之前的类似研究是 Mean-Flow based One-Step Vision-Language-Action 为CVPR2026。但这篇文章只使用了meanflow,我们的improved meanflow改进了meanflow中存在的一些理论问题,同时获得更稳定的训练效果,除此以外我们增加了attnres用于解决模型训练过程中的梯度消失和深层网络特征非常相似的问题。attnres的理论来源于大模型训练过程Residual connections [12] with PreNorm [60] are standard in modern LLMs, yet they accumulate
all layer outputs with fixed unit weights. This uniform aggregation causes uncontrolled hidden-state
growth with depth, progressively diluting each layers contribution [27]. We propose Attention
Residuals (AttnRes), which replaces this fixed accumulation with softmax attention over preceding
layer outputs, allowing each layer to selectively aggregate earlier representations with learned, input-
dependent weights. To address the memory and communication overhead of attending over all
preceding layer outputs for large-scale model training, we introduce Block AttnRes, which partitions
layers into blocks and attends over block-level representations, reducing the memory footprint while
preserving most of the gains of full AttnRes. Combined with cache-based pipeline communication
and a two-phase computation strategy, Block AttnRes becomes a practical drop-in replacement for
standard residual connections with minimal overhead.
Scaling law experiments confirm that the improvement is consistent across model sizes, and ablations
validate the benefit of content-dependent depth-wise selection. We further integrate AttnRes into
the Kimi Linear architecture [69] (48B total / 3B activated parameters) and pre-train on 1.4T tokens,
where AttnRes mitigates PreNorm dilution, yielding more uniform output magnitudes and gradient
distribution across depth, and improves downstream performance across all evaluated tasks 。我的概述如下 这里我们换另外一种写法,它能让我们看出更深刻的东西。先记 $\boldsymbol{y}_t = \boldsymbol{f}_t(\boldsymbol{x}_{t-1})$,那么有 $\boldsymbol{x}_t = \boldsymbol{x}_{t-1} + \boldsymbol{y}_t$,约定 $\boldsymbol{y}_0 = \boldsymbol{x}_0$,那么易得 $\boldsymbol{x}_t = \boldsymbol{y}_0 + \boldsymbol{y}_1 + \cdots + \boldsymbol{y}_t$,于是它可以等价地写成
$$
\boldsymbol{y}_{t+1} = \boldsymbol{f}_{t+1}(\boldsymbol{y}_0 + \boldsymbol{y}_1 + \cdots + \boldsymbol{y}_t) \tag{2}
$$
即从 $\boldsymbol{y}$ 的视角看,Residuals是将 $\boldsymbol{y}_0, \boldsymbol{y}_1, \cdots, \boldsymbol{y}_t$ 等权求和作为 $\boldsymbol{f}_{t+1}$ 的输入来得到 $\boldsymbol{y}_{t+1}$,那么一个自然的推广就是换成加权求和:
$$
\boldsymbol{y}_{t+1} = \boldsymbol{f}_{t+1}\left(\sum_{s=0}^t a_{t+1,s} \boldsymbol{y}_s\right) \qquad \text{where} \quad a_{t,s} \ge 0, \quad \sum_{s=0}^t a_{t+1,s} = 1 \tag{3}
$$
此时我们可以发现Hyper-connection就是这种加权求和的一种表现形式。但Hyper-connection中的$H$都由$tanh$激活后连乘得来,这导致该值有爆炸或者坍缩的风险。
Deepseek的mHC通过引入对H的交替归一化使其满足双随机性,双随机性即矩阵的所有行列和都为1,而且双随机性具有乘法的封闭性,即双随机矩阵相乘还是双随机矩阵,这样使得H的连乘不会出现爆炸和坍缩的问题。
回过头去考虑式3,既然是加权求和,那么一个权重的设置方法就是使用注意力机制,让权重$a$等于注意力分数,于是就有了AttnRes(苏神的作品)。其形式数学上看起来并不困难(RoPE也是这样)。
$$
a_{t+1,s} \propto \exp(\boldsymbol{w}_{t+1} \cdot \boldsymbol{y}_s) \tag{6}
$$
其中 $\boldsymbol{w}_t$ 是一个可训练的向量参数,即直接以一个数据无关的静态向量为Q、而K、V都是 $\boldsymbol{y}_s$ 去做Softmax Attention,这便是第一版AttnRes。随后通过一些实验得到给K多加个RMSNorm的操作,能取得比较稳定的收益,这构成了终版的AttnRes形式
$$
a_{t+1,s} \propto \exp(\boldsymbol{w}_{t+1} \cdot \text{RMSNorm}(\boldsymbol{y}_s)) \tag{7}
$$
我们研究用于对比的模型是Diffusion Policy,和ACT,在我自己的roboimi仿真环境下 有sim_transfer和pocket_insert两个仿真环境,在这个两个仿真环境下我使用imf-attnres和diffusion policy进行了很多实验。之前的实验都是使用swanlab记录的,实验以socket-insert开头 在runs目录下 可以看到之前的实验记录
+145
View File
@@ -0,0 +1,145 @@
\documentclass{article}
\usepackage{corl_2026} % Use this for the initial submission.
% \usepackage[final]{corl_2026} % Uncomment for the camera-ready ``final'' version.
% \usepackage[preprint]{corl_2026} % Uncomment for pre-prints (e.g., arxiv); This is like ``final'', but will remove the CORL footnote.
\title{Formatting Instructions for CoRL 2026 Camera-Ready}
% The \author macro works with any number of authors. There are two
% commands used to separate the names and addresses of multiple
% authors: \And and \AND.
%
% Using \And between authors leaves it to LaTeX to determine where to
% break the lines. Using \AND forces a line break at that point. So,
% if LaTeX puts 3 of 4 authors names on the first line, and the last
% on the second line, try using \AND instead of \And before the third
% author name.
% NOTE: authors will be visible only in the camera-ready and preprint versions (i.e., when using the option 'final' or 'preprint').
% For the initial submission the authors will be anonymized.
\author{
Jane E.~Doe\\
Department of Electrical Engineering and Computer Sciences\\
University of California Berkeley
United States\\
\texttt{janedoe@berkeley.edu} \\
%% examples of more authors
%% \And
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
%% \AND
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
%% \And
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
%% \And
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
}
\begin{document}
\maketitle
%===============================================================================
\begin{abstract}
The purpose of this document is to provide both the basic paper template and submission guidelines. Abstracts should be a single paragraph, between 4--6 sentences long, ideally. Gross violations will trigger corrections at the camera-ready phase.
\end{abstract}
% Two or three meaningful keywords should be added here
\keywords{CoRL, Robots, Learning}
%===============================================================================
\section{Introduction}
Submission to CoRL 2026 will be entirely electronic, via a web site (not email). Information about the submission process and \LaTeX{} templates are available on the conference web site at \url{https://corl.org/}. For camera ready submission, use the \texttt{final} option for the \texttt{\textbackslash usepackage} command.
%===============================================================================
\section{Citations}
\label{sec:citations}
Citations can be made using either \textbackslash citep\{\} or \textbackslash citet\{\}, depending from the appropriateness. To avoid the citation moving to the next line, it is often a good practice to replace the space before with a tilde (\~{}) character.
Example 1: ``CoRL is the best conference ever~\citep{Gauss1857}.''
Example 2: ``\citet{Lagrange1788} proved, both theoretically and numerically, that CoRL is the best conference ever.''
%===============================================================================
\section{Experimental Results}
\label{sec:result}
Nam dui ligula, fringilla a, euismod sodales, sollicitudin vel, wisi. Morbi auctor lorem non justo.
Nam lacus libero, pretium at, lobortis vitae, ultricies et, tellus. Donec aliquet, tortor sed accumsan
bibendum, erat ligula aliquet magna, vitae ornare odio metus a mi. Morbi ac orci et nisl hendrerit
mollis.
Suspendisse ut massa. Cras nec ante. Pellentesque a nulla. Cum sociis natoque penatibus
et magnis dis parturient montes, nascetur ridiculus mus. Aliquam tincidunt urna. Nulla ullamcorper
vestibulum turpis. Pellentesque cursus luctus mauris.
Nulla malesuada porttitor diam. Donec felis erat, congue non, volutpat at, tincidunt tristique, libero.
Vivamus viverra fermentum felis. Donec nonummy pellentesque ante. Phasellus adipiscing semper elit.
Proin fermentum massa ac quam. Sed diam turpis, molestie vitae, placerat a, molestie nec, leo.
Maecenas lacinia. Nam ipsum ligula, eleifend at, accumsan nec, suscipit a, ipsum. Morbi blandit
ligula feugiat magna. Nunc eleifend consequat lorem. Sed lacinia nulla vitae enim. Pellentesque tin-
cidunt purus vel magna. Integer non enim. Praesent euismod nunc eu purus. Donec bibendum quam
in tellus. Nullam cursus pulvinar lectus. Donec et mi. Nam vulputate metus eu enim. Vestibulum
pellentesque felis eu massa.
Quisque ullamcorper placerat ipsum. Cras nibh. Morbi vel justo vitae lacus tincidunt ultrices.
Lorem
ipsum dolor sit amet, consectetuer adipiscing elit. In hac habitasse platea dictumst. Integer tempus
convallis augue. Etiam facilisis. Nunc elementum fermentum wisi. Aenean placerat. Ut imperdiet,
enim sed gravida sollicitudin, felis odio placerat quam, ac pulvinar elit purus eget enim. Nunc vitae
tortor. Proin tempus nibh sit amet nisl. Vivamus quis tortor vitae risus porta vehicula.
%===============================================================================
\section{Conclusion}
\label{sec:conclusion}
Nam dui ligula, fringilla a, euismod sodales, sollicitudin vel, wisi. Morbi auctor lorem non justo.
Nam lacus libero, pretium at, lobortis vitae, ultricies et, tellus. Donec aliquet, tortor sed accumsan
bibendum, erat ligula aliquet magna, vitae ornare odio metus a mi. Morbi ac orci et nisl hendrerit
mollis. Suspendisse ut massa. Cras nec ante. Pellentesque a nulla. Cum sociis natoque penatibus
et magnis dis parturient montes, nascetur ridiculus mus. Aliquam tincidunt urna. Nulla ullamcorper
vestibulum turpis. Pellentesque cursus luctus mauris.
Nulla malesuada porttitor diam. Donec felis erat, congue non, volutpat at, tincidunt tristique, libero.
Vivamus viverra fermentum felis. Donec nonummy pellentesque ante. Phasellus adipiscing semper
elit. Proin fermentum massa ac quam. Sed diam turpis, molestie vitae, placerat a, molestie nec, leo.
Maecenas lacinia. Nam ipsum ligula, eleifend at, accumsan nec, suscipit a, ipsum. Morbi blandit
ligula feugiat magna. Nunc eleifend consequat lorem. Sed lacinia nulla vitae enim. Pellentesque tin-
cidunt purus vel magna. Integer non enim. Praesent euismod nunc eu purus. Donec bibendum quam
in tellus. Nullam cursus pulvinar lectus. Donec et mi. Nam vulputate metus eu enim. Vestibulum
pellentesque felis eu massa.
Quisque ullamcorper placerat ipsum. Cras nibh. Morbi vel justo vitae lacus tincidunt ultrices. Lorem
ipsum dolor sit amet, consectetuer adipiscing elit. In hac habitasse platea dictumst. Integer tempus
convallis augue. Etiam facilisis. Nunc elementum fermentum wisi. Aenean placerat. Ut imperdiet,
enim sed gravida sollicitudin, felis odio placerat quam, ac pulvinar elit purus eget enim. Nunc vitae
tortor. Proin tempus nibh sit amet nisl. Vivamus quis tortor vitae risus porta vehicula.
%===============================================================================
\clearpage
% The acknowledgments are automatically included only in the final and preprint versions of the paper.
\acknowledgments{If a paper is accepted, the final camera-ready version will (and probably should) include acknowledgments. All acknowledgments go at the end of the paper, including thanks to reviewers who gave useful comments, to colleagues who contributed to the ideas, and to funding agencies and corporate sponsors that provided financial support.}
%===============================================================================
% no \bibliographystyle is required, since the corl style is automatically used.
\bibliography{example} % .bib
\end{document}
@@ -0,0 +1,472 @@
% File: corl_2026.sty
%
% Latex templates for the Conference on Robot Learning (CoRL)
%
% This template is heavily inspired by the NeurIPS, ICML, ICLR and IEEE Transactions latex templates.
% Hence we would like to thank: Roman Garnett and the previous mantainers of the NIPS style, Percy Liang and the previous mantainers of the ICML style, Hugo Larochelle for the ICLR style, and Michael Shell for the IEEE Transactions style.
%
% History:
% 2017/04/16 - First revision by Roberto Calandra (roberto.calandra@berkeley.edu).
% Main changes:
% - The abstract is more compact compared to NeurIPS/ICML
% - References are by default using natbib with squared numbers (e.g., [1])
% - DOI fields from the bibtex are automatically converted to hyperlinks
% to the corresponding page
% - acknowledgments are now a command, and the corresponding subsubsection is
% automatically included only in the final version
% 2017/06/12 - Modified to use corlabbrvnat.bst, which order the reference by order of appearance in the paper
% 2017/06/13 - fixed typo
% 2018/05/09 - Slightly modified for CoRL 2018 by Jun Morimoto (xmorimo@atr.jp)
% 2019/01/28 - Slightly modified for CoRL 2019 by Jun Nakanishi (jnakanis@meijo-u.ac.jp)
% 2020/02/02 - Slightly modified for CoRL 2020 by Cynthia Matuszek (cmat@umbc.edu)
% 2020/08/19 - Added preprint option by Roberto Calandra (rcalandra@fb.com)
% 2021/05/06 - Slightly modified for CoRL 2021 by Gerhard Neumann (gerhard.neumann@kit.edu)
% 2022/03/09 - Slightly modified for CoRL 2022 by Minas Liarokapis (minas.liarokapis@auckland.ac.nz)
% 2022/03/06 - Slightly modified for CoRL 2023 by Marc Toussaint (toussaint@tu-berlin.de)
% 2024/03/26 - Slightly modified for CoRL 2024 by David Held (dheld@andrew.cmu.edu)
% 2026/01/10 - Slightly modified for CoRL 2026 by Yoonchang Sung (yoonchang.sung@ntu.edu.sg)
%
% TODO: nohyperref is not working at the moment
%
\NeedsTeXFormat{LaTeX2e}
% Content to be changed from year to year
\ProvidesPackage{corl_2026}[2026/08/15 CORL2026 submission/preprint/camera-ready style file]
\newcommand{\@conferenceordinal}{10th}
\newcommand{\@conferenceyear}{2026}
\newcommand{\@conferencelocation}{Austin TX, USA}
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Accepted options: [final,preprint,nonatbib,nohyperref]
% Declare the final option, which creates camera-ready copy
\newif\if@conferencefinal\@conferencefinalfalse
\DeclareOption{final}{
\@conferencefinaltrue
}
% Declare the preprint option, which creates a camera-ready copy without the corl footnote
\newif\if@preprinttype\@preprinttypefalse
\DeclareOption{preprint}{
\@preprinttypetrue
}
% The natbib package is loaded by default. Declaring the nonatbib option, does not load natbib in case of package clash (users can pass options to natbib via \PassOptionsToPackage)
\newif\if@natbib\@natbibtrue
\DeclareOption{nonatbib}{
\@natbibfalse
}
% The hyperref package is loaded by default. Declaring the nohyperref option, does not load the hyperref.
\DeclareOption{nohyperref}{%
\gdef\nohyperref{1}
}
% Activate the options
\ProcessOptions\relax
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
% Required packages:
\RequirePackage{lineno}
\RequirePackage{color}
% Load natbib unless told otherwise
\if@natbib
\RequirePackage[square,numbers]{natbib}
\bibliographystyle{corlabbrvnat}
\fi
% set page geometry
\RequirePackage{hyperref} % hyperlinks
\RequirePackage[verbose=true,letterpaper]{geometry}
\AtBeginDocument{
\newgeometry{
textheight=9in,
textwidth=5.5in,
top=1in,
headheight=12pt,
headsep=25pt,
footskip=30pt
}
\@ifpackageloaded{fullpage}
{\PackageWarning{corl_2026}{fullpage package not allowed! Overwriting formatting.}}
{}
}
\ifdefined\nohyperref\else\ifdefined\hypersetup
\definecolor{mydarkblue}{rgb}{0,0.08,0.45}
\hypersetup{ %
pdftitle={},
pdfauthor={},
pdfsubject={Proceedings of the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear)},
pdfkeywords={},
pdfborder=0 0 0,
pdfpagemode=UseNone,
colorlinks=true,
linkcolor=mydarkblue,
citecolor=mydarkblue,
filecolor=mydarkblue,
urlcolor=mydarkblue,
pdfview=FitH}
\ifdefined\isaccepted \else
\hypersetup{pdfauthor={Anonymous Submission}}
\fi
\fi\fi
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%
% fonts
\renewcommand{\rmdefault}{ptm}
\renewcommand{\sfdefault}{phv}
% Create acknowledgments -- only if the option 'final' is activated
\providecommand{\acknowledgments}{}
\renewcommand{\acknowledgments}[1]{%
\if@conferencefinal%
\subsubsection*{Acknowledgments} #1
\fi
\if@preprinttype%
\subsubsection*{Acknowledgments} #1
\fi
}
% handle tweaks for camera-ready copy vs. submission copy
\if@conferencefinal
\newcommand{\@noticestring}{%
\@conferenceordinal\/ Conference on Robot Learning
(CoRL \@conferenceyear), \@conferencelocation.%
}
\else
\if@preprinttype
\newcommand{\@noticestring}{%
% Nothing here.
}
\else
\newcommand{\@noticestring}{%
Submitted to the \@conferenceordinal\/ Conference on Robot Learning (CoRL \@conferenceyear). Do not distribute.%
}
% line numbers for submission
\linenumbers
% fix incompatibilities between lineno and amsmath, if required, by
% transparently wrapping linenomath environments around amsmath
% environments
\AtBeginDocument{%
\@ifpackageloaded{amsmath}{%
\newcommand*\patchAmsMathEnvironmentForLineno[1]{%
\expandafter\let\csname old#1\expandafter\endcsname\csname #1\endcsname
\expandafter\let\csname oldend#1\expandafter\endcsname\csname end#1\endcsname
\renewenvironment{#1}%
{\linenomath\csname old#1\endcsname}%
{\csname oldend#1\endcsname\endlinenomath}%
}%
\newcommand*\patchBothAmsMathEnvironmentsForLineno[1]{%
\patchAmsMathEnvironmentForLineno{#1}%
\patchAmsMathEnvironmentForLineno{#1*}%
}%
\patchBothAmsMathEnvironmentsForLineno{equation}%
\patchBothAmsMathEnvironmentsForLineno{align}%
\patchBothAmsMathEnvironmentsForLineno{flalign}%
\patchBothAmsMathEnvironmentsForLineno{alignat}%
\patchBothAmsMathEnvironmentsForLineno{gather}%
\patchBothAmsMathEnvironmentsForLineno{multline}%
}{}
}
\fi
\fi
% The DOI will now automatically generate a URL link. Pretty cool!
% + Fix for hyperref and DOI (https://www.tug.org/pipermail/tex-live/2012-August/032161.html)
%-----------------------------
\makeatletter
\providecommand{\doi}[1]{%
\begingroup
\let\bibinfo\@secondoftwo
\urlstyle{rm}%
\href{http://dx.doi.org/#1}{%
doi:\discretionary{}{}{}%
\nolinkurl{#1}%
}%
\endgroup
}
% \makeatother
% %-----------------------------
\widowpenalty=10000
\clubpenalty=10000
\flushbottom
\sloppy
% font sizes with reduced leading
\renewcommand{\normalsize}{%
\@setfontsize\normalsize\@xpt\@xipt
\abovedisplayskip 7\p@ \@plus 2\p@ \@minus 5\p@
\abovedisplayshortskip \z@ \@plus 3\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 4\p@ \@plus 3\p@ \@minus 3\p@
}
\normalsize
\renewcommand{\small}{%
\@setfontsize\small\@ixpt\@xpt
\abovedisplayskip 6\p@ \@plus 1.5\p@ \@minus 4\p@
\abovedisplayshortskip \z@ \@plus 2\p@
\belowdisplayskip \abovedisplayskip
\belowdisplayshortskip 3\p@ \@plus 2\p@ \@minus 2\p@
}
\renewcommand{\footnotesize}{\@setfontsize\footnotesize\@ixpt\@xpt}
\renewcommand{\scriptsize}{\@setfontsize\scriptsize\@viipt\@viiipt}
\renewcommand{\tiny}{\@setfontsize\tiny\@vipt\@viipt}
\renewcommand{\large}{\@setfontsize\large\@xiipt{14}}
\renewcommand{\Large}{\@setfontsize\Large\@xivpt{16}}
\renewcommand{\LARGE}{\@setfontsize\LARGE\@xviipt{20}}
\renewcommand{\huge}{\@setfontsize\huge\@xxpt{23}}
\renewcommand{\Huge}{\@setfontsize\Huge\@xxvpt{28}}
% sections with less space
\providecommand{\section}{}
\renewcommand{\section}{%
\@startsection{section}{1}{\z@}%
{-2.0ex \@plus -0.5ex \@minus -0.2ex}%
{ 1.5ex \@plus 0.3ex \@minus 0.2ex}%
{\large\bf\raggedright}%
}
\providecommand{\subsection}{}
\renewcommand{\subsection}{%
\@startsection{subsection}{2}{\z@}%
{-1.8ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.8ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\subsubsection}{}
\renewcommand{\subsubsection}{%
\@startsection{subsubsection}{3}{\z@}%
{-1.5ex \@plus -0.5ex \@minus -0.2ex}%
{ 0.5ex \@plus 0.2ex}%
{\normalsize\bf\raggedright}%
}
\providecommand{\paragraph}{}
\renewcommand{\paragraph}{%
\@startsection{paragraph}{4}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subparagraph}{}
\renewcommand{\subparagraph}{%
\@startsection{subparagraph}{5}{\z@}%
{1.5ex \@plus 0.5ex \@minus 0.2ex}%
{-1em}%
{\normalsize\bf}%
}
\providecommand{\subsubsubsection}{}
\renewcommand{\subsubsubsection}{%
\vskip5pt{\noindent\normalsize\rm\raggedright}%
}
% float placement
\renewcommand{\topfraction }{0.85}
\renewcommand{\bottomfraction }{0.4}
\renewcommand{\textfraction }{0.1}
\renewcommand{\floatpagefraction}{0.7}
\newlength{\@nipsabovecaptionskip}\setlength{\@nipsabovecaptionskip}{7\p@}
\newlength{\@nipsbelowcaptionskip}\setlength{\@nipsbelowcaptionskip}{\z@}
\setlength{\abovecaptionskip}{\@nipsabovecaptionskip}
\setlength{\belowcaptionskip}{\@nipsbelowcaptionskip}
% swap above/belowcaptionskip lengths for tables
\renewenvironment{table}
{\setlength{\abovecaptionskip}{\@nipsbelowcaptionskip}%
\setlength{\belowcaptionskip}{\@nipsabovecaptionskip}%
\@float{table}}
{\end@float}
% footnote formatting
\setlength{\footnotesep }{6.65\p@}
\setlength{\skip\footins}{9\p@ \@plus 4\p@ \@minus 2\p@}
\renewcommand{\footnoterule}{\kern-3\p@ \hrule width 12pc \kern 2.6\p@}
\setcounter{footnote}{0}
% paragraph formatting
\setlength{\parindent}{\z@}
\setlength{\parskip }{5.5\p@}
% list formatting
\setlength{\topsep }{4\p@ \@plus 1\p@ \@minus 2\p@}
\setlength{\partopsep }{1\p@ \@plus 0.5\p@ \@minus 0.5\p@}
\setlength{\itemsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\parsep }{2\p@ \@plus 1\p@ \@minus 0.5\p@}
\setlength{\leftmargin }{3pc}
\setlength{\leftmargini }{\leftmargin}
\setlength{\leftmarginii }{2em}
\setlength{\leftmarginiii}{1.5em}
\setlength{\leftmarginiv }{1.0em}
\setlength{\leftmarginv }{0.5em}
\def\@listi {\leftmargin\leftmargini}
\def\@listii {\leftmargin\leftmarginii
\labelwidth\leftmarginii
\advance\labelwidth-\labelsep
\topsep 2\p@ \@plus 1\p@ \@minus 0.5\p@
\parsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\itemsep \parsep}
\def\@listiii{\leftmargin\leftmarginiii
\labelwidth\leftmarginiii
\advance\labelwidth-\labelsep
\topsep 1\p@ \@plus 0.5\p@ \@minus 0.5\p@
\parsep \z@
\partopsep 0.5\p@ \@plus 0\p@ \@minus 0.5\p@
\itemsep \topsep}
\def\@listiv {\leftmargin\leftmarginiv
\labelwidth\leftmarginiv
\advance\labelwidth-\labelsep}
\def\@listv {\leftmargin\leftmarginv
\labelwidth\leftmarginv
\advance\labelwidth-\labelsep}
\def\@listvi {\leftmargin\leftmarginvi
\labelwidth\leftmarginvi
\advance\labelwidth-\labelsep}
% create title
\providecommand{\maketitle}{}
\renewcommand{\maketitle}{%
\par
\begingroup
\renewcommand{\thefootnote}{\fnsymbol{footnote}}
% for perfect author name centering
\renewcommand{\@makefnmark}{\hbox to \z@{$^{\@thefnmark}$\hss}}
% The footnote-mark was overlapping the footnote-text,
% added the following to fix this problem (MK)
\long\def\@makefntext##1{%
\parindent 1em\noindent
\hbox to 1.8em{\hss $\m@th ^{\@thefnmark}$}##1
}
\thispagestyle{empty}
\@maketitle
\@thanks
\@notice
\endgroup
\let\maketitle\relax
\let\thanks\relax
}
% rules for title box at top of first page
\newcommand{\@toptitlebar}{
\hrule height 4\p@
\vskip 0.25in
\vskip -\parskip%
}
\newcommand{\@bottomtitlebar}{
\vskip 0.29in
\vskip -\parskip
\hrule height 1\p@
\vskip 0.09in%
}
%% keywords as first class citizens
\def\keywords#1{%
% \ifdefined\isaccepted \else
% \par {\bf Keywords:} #1%
% \fi
% \ifdefined\nohyperref\else\ifdefined\hypersetup
% \hypersetup{pdfkeywords={#1}}
% \fi\fi
\ifdefined\isaccepted \else
\begin{quote}
\textbf{Keywords:} #1%
\end{quote}
\fi
\ifdefined\nohyperref\else\ifdefined\hypersetup
\hypersetup{pdfkeywords={#1}}
\fi\fi
}
% create title (includes both anonymized and non-anonymized versions)
\providecommand{\@maketitle}{}
\renewcommand{\@maketitle}{%
\vbox{%
\hsize\textwidth
\linewidth\hsize
\vskip 0.1in
% \@toptitlebar
\centering
{\LARGE\bf \@title\par}
% \@bottomtitlebar
\if@conferencefinal
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\if@preprinttype
\def\And{%
\end{tabular}\hfil\linebreak[0]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\def\AND{%
\end{tabular}\hfil\linebreak[4]\hfil%
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\ignorespaces%
}
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}\@author\end{tabular}%
\else
\begin{tabular}[t]{c}\bf\rule{\z@}{24\p@}
Anonymous Author(s) \\
Affiliation \\
Address \\
\texttt{email} \\
\end{tabular}%
\fi
\fi
\vskip 0.3in \@minus 0.1in
}
}
% add conference notice to bottom of first page
\newcommand{\ftype@noticebox}{8}
\newcommand{\@notice}{%
% give a bit of extra room back to authors on first page
\enlargethispage{2\baselineskip}%
\@float{noticebox}[b]%
\footnotesize\@noticestring%
\end@float%
}
% abstract styling
\renewenvironment{abstract}%
{%
% \vskip 0.075in%
% \centerline%
% {\large\bf Abstract}%
% \vspace{0.5ex}%
\begin{quote}%
\textbf{Abstract:}%
}
{
\par%
\vskip 1ex%
% \ifdefined\keywords
% \textbf{Keywords:}%
% \@keywords%
% \else
% \fi
\end{quote}%
}
\endinput
@@ -0,0 +1,20 @@
% This file was created with JabRef 2.10.
% Encoding: UTF-8
@Article{Gauss1857,
Title = {Theory of the motion of the heavenly bodies moving about the sun in conic sections},
Author = {Carl Friedrich Gauss and Charles Henry Davis},
Journal = {Gauss's Theoria Motus},
Year = {1857},
Number = {1},
Pages = {5--23},
Volume = {76}
}
@Book{Lagrange1788,
title = {M{\'e}canique Analytique},
author = {Joseph-Louis Lagrange},
publisher = {Desaint, Paris},
year = {1788}
}
@@ -0,0 +1,145 @@
\documentclass{article}
\usepackage{corl_2026} % Use this for the initial submission.
% \usepackage[final]{corl_2026} % Uncomment for the camera-ready ``final'' version.
% \usepackage[preprint]{corl_2026} % Uncomment for pre-prints (e.g., arxiv); This is like ``final'', but will remove the CORL footnote.
\title{Formatting Instructions for CoRL 2026 Camera-Ready}
% The \author macro works with any number of authors. There are two
% commands used to separate the names and addresses of multiple
% authors: \And and \AND.
%
% Using \And between authors leaves it to LaTeX to determine where to
% break the lines. Using \AND forces a line break at that point. So,
% if LaTeX puts 3 of 4 authors names on the first line, and the last
% on the second line, try using \AND instead of \And before the third
% author name.
% NOTE: authors will be visible only in the camera-ready and preprint versions (i.e., when using the option 'final' or 'preprint').
% For the initial submission the authors will be anonymized.
\author{
Jane E.~Doe\\
Department of Electrical Engineering and Computer Sciences\\
University of California Berkeley
United States\\
\texttt{janedoe@berkeley.edu} \\
%% examples of more authors
%% \And
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
%% \AND
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
%% \And
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
%% \And
%% Coauthor \\
%% Affiliation \\
%% Address \\
%% \texttt{email} \\
}
\begin{document}
\maketitle
%===============================================================================
\begin{abstract}
The purpose of this document is to provide both the basic paper template and submission guidelines. Abstracts should be a single paragraph, between 4--6 sentences long, ideally. Gross violations will trigger corrections at the camera-ready phase.
\end{abstract}
% Two or three meaningful keywords should be added here
\keywords{CoRL, Robots, Learning}
%===============================================================================
\section{Introduction}
Submission to CoRL 2026 will be entirely electronic, via a web site (not email). Information about the submission process and \LaTeX{} templates are available on the conference web site at \url{https://corl.org/}. For camera ready submission, use the \texttt{final} option for the \texttt{\textbackslash usepackage} command.
%===============================================================================
\section{Citations}
\label{sec:citations}
Citations can be made using either \textbackslash citep\{\} or \textbackslash citet\{\}, depending from the appropriateness. To avoid the citation moving to the next line, it is often a good practice to replace the space before with a tilde (\~{}) character.
Example 1: ``CoRL is the best conference ever~\citep{Gauss1857}.''
Example 2: ``\citet{Lagrange1788} proved, both theoretically and numerically, that CoRL is the best conference ever.''
%===============================================================================
\section{Experimental Results}
\label{sec:result}
Nam dui ligula, fringilla a, euismod sodales, sollicitudin vel, wisi. Morbi auctor lorem non justo.
Nam lacus libero, pretium at, lobortis vitae, ultricies et, tellus. Donec aliquet, tortor sed accumsan
bibendum, erat ligula aliquet magna, vitae ornare odio metus a mi. Morbi ac orci et nisl hendrerit
mollis.
Suspendisse ut massa. Cras nec ante. Pellentesque a nulla. Cum sociis natoque penatibus
et magnis dis parturient montes, nascetur ridiculus mus. Aliquam tincidunt urna. Nulla ullamcorper
vestibulum turpis. Pellentesque cursus luctus mauris.
Nulla malesuada porttitor diam. Donec felis erat, congue non, volutpat at, tincidunt tristique, libero.
Vivamus viverra fermentum felis. Donec nonummy pellentesque ante. Phasellus adipiscing semper elit.
Proin fermentum massa ac quam. Sed diam turpis, molestie vitae, placerat a, molestie nec, leo.
Maecenas lacinia. Nam ipsum ligula, eleifend at, accumsan nec, suscipit a, ipsum. Morbi blandit
ligula feugiat magna. Nunc eleifend consequat lorem. Sed lacinia nulla vitae enim. Pellentesque tin-
cidunt purus vel magna. Integer non enim. Praesent euismod nunc eu purus. Donec bibendum quam
in tellus. Nullam cursus pulvinar lectus. Donec et mi. Nam vulputate metus eu enim. Vestibulum
pellentesque felis eu massa.
Quisque ullamcorper placerat ipsum. Cras nibh. Morbi vel justo vitae lacus tincidunt ultrices.
Lorem
ipsum dolor sit amet, consectetuer adipiscing elit. In hac habitasse platea dictumst. Integer tempus
convallis augue. Etiam facilisis. Nunc elementum fermentum wisi. Aenean placerat. Ut imperdiet,
enim sed gravida sollicitudin, felis odio placerat quam, ac pulvinar elit purus eget enim. Nunc vitae
tortor. Proin tempus nibh sit amet nisl. Vivamus quis tortor vitae risus porta vehicula.
%===============================================================================
\section{Conclusion}
\label{sec:conclusion}
Nam dui ligula, fringilla a, euismod sodales, sollicitudin vel, wisi. Morbi auctor lorem non justo.
Nam lacus libero, pretium at, lobortis vitae, ultricies et, tellus. Donec aliquet, tortor sed accumsan
bibendum, erat ligula aliquet magna, vitae ornare odio metus a mi. Morbi ac orci et nisl hendrerit
mollis. Suspendisse ut massa. Cras nec ante. Pellentesque a nulla. Cum sociis natoque penatibus
et magnis dis parturient montes, nascetur ridiculus mus. Aliquam tincidunt urna. Nulla ullamcorper
vestibulum turpis. Pellentesque cursus luctus mauris.
Nulla malesuada porttitor diam. Donec felis erat, congue non, volutpat at, tincidunt tristique, libero.
Vivamus viverra fermentum felis. Donec nonummy pellentesque ante. Phasellus adipiscing semper
elit. Proin fermentum massa ac quam. Sed diam turpis, molestie vitae, placerat a, molestie nec, leo.
Maecenas lacinia. Nam ipsum ligula, eleifend at, accumsan nec, suscipit a, ipsum. Morbi blandit
ligula feugiat magna. Nunc eleifend consequat lorem. Sed lacinia nulla vitae enim. Pellentesque tin-
cidunt purus vel magna. Integer non enim. Praesent euismod nunc eu purus. Donec bibendum quam
in tellus. Nullam cursus pulvinar lectus. Donec et mi. Nam vulputate metus eu enim. Vestibulum
pellentesque felis eu massa.
Quisque ullamcorper placerat ipsum. Cras nibh. Morbi vel justo vitae lacus tincidunt ultrices. Lorem
ipsum dolor sit amet, consectetuer adipiscing elit. In hac habitasse platea dictumst. Integer tempus
convallis augue. Etiam facilisis. Nunc elementum fermentum wisi. Aenean placerat. Ut imperdiet,
enim sed gravida sollicitudin, felis odio placerat quam, ac pulvinar elit purus eget enim. Nunc vitae
tortor. Proin tempus nibh sit amet nisl. Vivamus quis tortor vitae risus porta vehicula.
%===============================================================================
\clearpage
% The acknowledgments are automatically included only in the final and preprint versions of the paper.
\acknowledgments{If a paper is accepted, the final camera-ready version will (and probably should) include acknowledgments. All acknowledgments go at the end of the paper, including thanks to reviewers who gave useful comments, to colleagues who contributed to the ideas, and to funding agencies and corporate sponsors that provided financial support.}
%===============================================================================
% no \bibliographystyle is required, since the corl style is automatically used.
\bibliography{example} % .bib
\end{document}
+398
View File
@@ -0,0 +1,398 @@
{
"tables": [
{
"label": "Socket peg insertion, 100-rollout performance and speed",
"headers": [
"run_name",
"method_family",
"horizon_or_steps",
"avg_reward",
"median_reward",
"avg_max_reward",
"max_reward",
"nonzero_reward",
"success_like",
"avg_inference_fps",
"avg_control_fps",
"avg_inference_time_ms"
],
"rows": [
[
"socket-insert-imf-attnres-ph32-exec16-emb384-l12-infer1-unfreeze-step50k-roll1x10-5880g01-20260506-195806",
"iMF-AttnRes",
"infer1, ph32, exec16, 50k",
"1275.22",
"1490.5",
"3.12",
"2250.0",
"84/100",
"0/100",
"64.226",
"5.863",
"16.186"
],
[
"socket-insert-no-pretrain-ph16-emb384-l18-infer100-unfreeze-step150k-roll1x10-5090-20260506-193753",
"Diffusion-style VLA",
"infer100, ph16, 150k",
"861.53",
"668.0",
"2.58",
"2241.0",
"97/100",
"2/100",
"2.956",
"2.591",
"338.291"
],
[
"socket-insert-no-pretrain-ph32-exec16-emb384-l18-infer100-unfreeze-step150k-roll1x10-5880g1-20260507-092227",
"Diffusion-style VLA",
"infer100, ph32, exec16, 150k",
"1019.39",
"804.0",
"2.47",
"2385.0",
"92/100",
"4/100",
"2.516",
"1.928",
"397.912"
],
[
"socket-insert-imf-attnres-ph32-exec16-emb384-l12-infer2-drop005-unfreeze-step150k-roll1x10-5880g01-20260508-165022",
"iMF-AttnRes",
"infer2, ph32, exec16, 150k",
"1472.62",
"1825.0",
"3.29",
"2365.0",
"83/100",
"5/100",
"71.579",
"7.181",
"14.409"
],
[
"socket-insert-imf-attnres-ph32-exec16-emb384-l12-infer3-drop005-unfreeze-step150k-roll1x10-5880g01-20260509-172423",
"iMF-AttnRes",
"infer3, ph32, exec16, 150k",
"1513.56",
"1901.5",
"3.28",
"2285.0",
"83/100",
"3/100",
"67.891",
"7.247",
"15.122"
],
[
"act-socket-peg-224-20260508-170237",
"ACT",
"action chunking",
"289.6",
"13.0",
"1.29",
"2091.0",
"55/100",
"1/100",
"10.5171",
"2.2782",
"100.2116"
],
[
"smolvla_socket_peg_bs80_100k_20260511_145727",
"SmolVLA",
"100k",
"466.16",
"106.0",
"1.72",
"4.0",
"89/100",
"0/100",
"382.53",
"16.84",
"2.614"
]
]
},
{
"label": "Object transfer / sim_transfer, 100-rollout performance and speed",
"headers": [
"run_name",
"method_family",
"horizon_or_steps",
"avg_reward",
"median_reward",
"avg_max_reward",
"max_reward",
"nonzero_reward",
"success_like",
"avg_inference_fps",
"avg_control_fps",
"avg_inference_time_ms"
],
"rows": [
[
"diffusion_policy_native_dit_ddpm_resnet_best",
"Diffusion Policy native DiT + DDPM + ResNet",
"best checkpoint",
"319.2",
"6.0",
"1.68",
"1416.0",
"55/100",
"29/100",
"32.09",
"16.70",
"n/a"
],
[
"embed384_layer18_best_checkpoint",
"Diffusion-style VLA",
"emb384, layer18",
"233.52",
"0.0",
"1.12",
"1436.0",
"39/100",
"17/100",
"1.859",
"1.649",
"n/a"
],
[
"resnet18-multitoken-imf-emb256-l16-ph16-ex08-roll10-l20g2-20260406-112815",
"iMF multi-token ResNet18",
"step34999, ph16, exec08",
"260.66",
"0.0",
"1.12",
"1422.0",
"33/100",
"23/100",
"441.80",
"13.37",
"n/a"
],
[
"imf-p2-full-attnres-vision-ph16-ex08-emb384-l12-b40-lr1p25e4-ms50k-l20g3-20260405-002424",
"iMF full AttnRes vision",
"ph16, exec08, 50k",
"228.42",
"0.0",
"0.94",
"1482.0",
"31/100",
"16/100",
"55.997",
"4.301",
"n/a"
],
[
"imf-p1-ph16-ex16-emb384-l12-ms50k-l20g0-20260404-131223",
"iMF-AttnRes DiT only",
"ph16, exec16, 50k",
"240.64",
"0.0",
"1.02",
"1428.0",
"32/100",
"19/100",
"137.996",
"4.523",
"n/a"
],
[
"imf-p1-ph32-ex08-emb384-l12-ms50k-l20g1-20260404-131223",
"iMF-AttnRes DiT only",
"ph32, exec08, 50k",
"163.28",
"0.0",
"0.76",
"1566.0",
"26/100",
"12/100",
"85.638",
"4.317",
"n/a"
],
[
"imf-p1-ph32-ex32-emb384-l12-ms50k-5090-20260404-13122",
"iMF-AttnRes DiT only",
"ph32, exec32, 50k",
"260.72",
"0.0",
"1.18",
"1342.0",
"38/100",
"21/100",
"9.557",
"3.299",
"n/a"
],
[
"imf-p1-ph16-ex08-emb384-l12-ms50k-5880g1-20260404-131223",
"iMF-AttnRes DiT only",
"ph16, exec08, 50k",
"229.56",
"0.0",
"1.18",
"1564.0",
"41/100",
"18/100",
"69.984",
"8.053",
"n/a"
],
[
"imf-p1-ph08-ex08-emb384-l12-ms50k-5880g0-20260404-131223",
"iMF-AttnRes DiT only",
"ph08, exec08, 50k",
"237.02",
"0.0",
"1.32",
"1400.0",
"48/100",
"18/100",
"69.192",
"8.173",
"n/a"
],
[
"imf-p1-ph32-ex16-emb384-l12-ms50k-l20g2-20260404-131223",
"iMF-AttnRes DiT only",
"ph32, exec16, 50k",
"49.88",
"0.0",
"0.32",
"1284.0",
"13/100",
"4/100",
"138.903",
"4.482",
"n/a"
],
[
"sim-transfer-imf-attnres-ph32-exec16-emb384-l12-infer1-unfreeze-step50k-roll5x5-20260403-14130",
"iMF-AttnRes",
"infer1, ph32, exec16, 50k",
"526.22",
"86.0",
"2.14",
"1460.0",
"63/100",
"44/100",
"8.911",
"3.193",
"n/a"
]
]
},
{
"label": "Derived comparisons used in the paper draft",
"headers": [
"comparison",
"metric",
"numerator_run",
"denominator_run",
"numerator_value",
"denominator_value",
"derived_ratio_or_delta"
],
"rows": [
[
"socket best iMF vs ph16 diffusion baseline",
"avg_reward ratio",
"iMF infer3 150k",
"diffusion ph16 infer100 150k",
"1513.56",
"861.53",
"1.76x"
],
[
"socket best iMF vs ph32 diffusion baseline",
"avg_reward ratio",
"iMF infer3 150k",
"diffusion ph32 infer100 150k",
"1513.56",
"1019.39",
"1.48x"
],
[
"socket best iMF vs ACT",
"avg_reward ratio",
"iMF infer3 150k",
"ACT",
"1513.56",
"289.6",
"5.23x"
],
[
"socket best iMF vs SmolVLA",
"avg_reward ratio",
"iMF infer3 150k",
"SmolVLA",
"1513.56",
"466.16",
"3.25x"
],
[
"socket iMF infer2 vs ph16 diffusion baseline",
"inference time reduction",
"iMF infer2 150k",
"diffusion ph16 infer100 150k",
"14.409",
"338.291",
"23.48x lower latency"
],
[
"socket iMF infer3 vs ph16 diffusion baseline",
"inference fps ratio",
"iMF infer3 150k",
"diffusion ph16 infer100 150k",
"67.891",
"2.956",
"22.97x"
],
[
"socket iMF infer2 vs ph32 diffusion baseline",
"inference fps ratio",
"iMF infer2 150k",
"diffusion ph32 infer100 150k",
"71.579",
"2.516",
"28.45x"
],
[
"sim_transfer best iMF vs native diffusion policy",
"avg_reward ratio",
"sim-transfer iMF infer1",
"native diffusion policy",
"526.22",
"319.2",
"1.65x"
],
[
"sim_transfer best iMF vs native diffusion policy",
"success_like delta",
"sim-transfer iMF infer1",
"native diffusion policy",
"44",
"29",
"+15 episodes"
],
[
"sim_transfer ResNet18 multitoken iMF vs embed384 layer18 baseline",
"inference fps ratio",
"ResNet18 multitoken iMF",
"embed384 layer18",
"441.80",
"1.859",
"237.65x"
]
]
}
]
}
+369
View File
@@ -0,0 +1,369 @@
{
"plotting_plan": [
{
"figure_id": "fig_method_overview",
"title": "iMF-AttnRes VLA Overview",
"plot_type": "diagram",
"data_source": "idea.md",
"objective": "Paper Banana prompt placeholder for a conceptual pipeline diagram showing visual observations and robot state entering a VLA policy, iMF predicting average flow with one-to-few inference evaluations, AttnRes aggregating depth-wise residual features, and chunked robot actions being executed in RoboIMI.",
"aspect_ratio": "16:9"
},
{
"figure_id": "fig_socket_reward_latency",
"title": "Socket Insertion Reward-Latency Tradeoff",
"plot_type": "plot",
"data_source": "experimental_log.md",
"objective": "Paper Banana prompt placeholder for a scatter/bar hybrid plot comparing socket-insert avg_reward against avg_inference_time_ms or inference FPS for iMF-AttnRes, diffusion-style VLA, ACT, and SmolVLA, emphasizing that iMF-AttnRes reaches higher reward than diffusion-style baselines while using roughly 14--16 ms inference latency instead of hundreds of milliseconds.",
"aspect_ratio": "4:3"
},
{
"figure_id": "fig_sim_transfer_ablation",
"title": "Sim Transfer Ablation Summary",
"plot_type": "plot",
"data_source": "experimental_log.md",
"objective": "Paper Banana prompt placeholder for a grouped bar chart over sim_transfer ablations showing avg_reward and success_like episode count for native diffusion policy, best iMF-AttnRes, multi-token ResNet18 iMF, full-AttnRes vision, and horizon/execution variants.",
"aspect_ratio": "4:3"
},
{
"figure_id": "fig_imf_attnres_components",
"title": "iMF and AttnRes Component Mechanics",
"plot_type": "diagram",
"data_source": "idea.md",
"objective": "Paper Banana prompt placeholder for a technical diagram contrasting classical flow matching instantaneous velocity prediction, Mean Flow average velocity prediction, iMF's corrected instantaneous-velocity training relation, and AttnRes softmax aggregation over preceding residual outputs.",
"aspect_ratio": "16:9"
}
],
"intro_related_work_plan": {
"introduction_strategy": {
"hook_hypothesis": "Diffusion-style action generators can produce expressive robot policies, but their iterative denoising or ODE-style sampling creates a latency bottleneck for closed-loop robot control.",
"problem_gap_hypothesis": "Existing fast robot policies often trade away action-distribution expressivity or require distillation, while existing one-step flow methods have not fully explored improved mean-flow training and depth-wise residual aggregation in VLA imitation learning.",
"search_directions": [
"Find foundational and survey papers on imitation learning and visuomotor policy learning for robotic manipulation before 2026-04-15",
"Find papers on diffusion policy and action diffusion for visuomotor robot control before 2026-04-15",
"Find papers on vision-language-action robot policies, including RT-1, RT-2, OpenVLA, SmolVLA, and flow-based VLA models before 2026-04-15",
"Find papers on flow matching, rectified flow, Mean Flow, and improved Mean Flow for one-step or few-step generation before 2026-04-15",
"Find papers on residual connections, Transformers, PreNorm residual architectures, hyper-connections, and attention residual aggregation before 2026-04-15"
]
},
"related_work_strategy": {
"overview": "Organize related work around robot action generation, fast flow-based generation, and residual/attention architecture design. The review should explain why a one-to-few-step iMF policy with AttnRes is a plausible alternative to iterative diffusion action generation.",
"subsections": [
{
"subsection_title": "2.1 Diffusion and Flow Policies for Robot Action Generation",
"methodology_cluster": "Diffusion Policy, action diffusion, flow matching, and generative action policies",
"sota_investigation_mission": "Identify influential robot manipulation policies that use diffusion, flow matching, or iterative generative sampling, and summarize their sampling-step and latency implications.",
"limitation_hypothesis": "Iterative sampling improves expressivity but makes high-frequency closed-loop control expensive, especially when policies require tens or hundreds of inference evaluations per action chunk.",
"limitation_search_queries": [
"Diffusion Policy Visuomotor Policy Learning via Action Diffusion latency sampling steps",
"robot manipulation action diffusion policy inference speed sampling steps",
"flow matching robot policy action generation one-step few-step"
],
"bridge_to_our_method": "iMF-AttnRes targets the same expressive action-generation setting but trains the policy to predict an average flow usable with one-to-few inference evaluations."
},
{
"subsection_title": "2.2 Vision-Language-Action Models and Action Chunking",
"methodology_cluster": "VLA policies, action tokenization, action chunking, and transformer-based robot policies",
"sota_investigation_mission": "Survey VLA and transformer imitation-learning systems that condition robot actions on language, images, and robot state, including models that improve deployment speed via chunking or compact action representations.",
"limitation_hypothesis": "Compact VLA and chunking approaches can improve speed, but may not directly address the multi-step generative sampling cost of diffusion-style policies or the training instability of naive one-step flow policies.",
"limitation_search_queries": [
"OpenVLA vision language action robot policy imitation learning",
"RT-1 RT-2 robotics transformer vision language action control",
"Action Chunking with Transformers ALOHA imitation learning",
"SmolVLA efficient vision language action robotics"
],
"bridge_to_our_method": "The proposed policy keeps the VLA imitation-learning framing while replacing slow iterative action sampling with iMF and strengthening deep policy representation with AttnRes."
},
{
"subsection_title": "2.3 Mean-Flow Training for One-Step Generation",
"methodology_cluster": "Flow matching, rectified flow, Mean Flow, improved Mean Flow, and fast-forward generative models",
"sota_investigation_mission": "Trace the development from flow matching and rectified flow to Mean Flow and improved Mean Flow, focusing on how average velocity prediction enables one-step or few-step generation.",
"limitation_hypothesis": "Naively substituting conditional velocities into nonlinear mean-flow training relations can introduce unstable objectives; improved Mean Flow addresses this with a corrected training relation that better preserves flow-matching stability.",
"limitation_search_queries": [
"Mean Flows for One-step Generative Modeling average velocity",
"Improved Mean Flows fast-forward generative models conditional velocity marginal velocity",
"Flow Matching for Generative Modeling conditional vector field",
"Rectified Flow generative modeling one step sampling"
],
"bridge_to_our_method": "We adapt the improved mean-flow idea from image generation to robot action generation, where lower sampling latency directly affects control frequency."
},
{
"subsection_title": "2.4 Residual Aggregation and Attention Residuals in Deep Transformers",
"methodology_cluster": "Residual connections, PreNorm Transformers, hyper-connections, and AttnRes",
"sota_investigation_mission": "Investigate why residual paths are central to deep networks and how learned or attention-based residual aggregation can mitigate feature dilution and gradient imbalance in deep transformers.",
"limitation_hypothesis": "Standard unit-weight residual accumulation treats all previous layer outputs equally, which can dilute individual layer contributions and produce depth-wise representation redundancy in deep VLA policies.",
"limitation_search_queries": [
"Attention Residuals AttnRes PreNorm dilution hidden-state growth",
"residual connections transformer PreNorm gradient flow depth",
"hyper-connections deep transformers residual aggregation",
"Deep Residual Learning for Image Recognition residual connections"
],
"bridge_to_our_method": "AttnRes supplies a drop-in way for each policy layer to selectively aggregate earlier residual outputs rather than relying on fixed unit-weight accumulation."
}
]
}
},
"section_plan": [
{
"section_title": "Abstract",
"subsections": [
{
"subsection_title": "Abstract Content",
"content_bullets": [
"State the robotics problem: diffusion-style VLA policies are expressive but slow when closed-loop control needs many policy evaluations.",
"Introduce iMF-AttnRes as a VLA imitation-learning policy combining improved Mean Flow for one-to-few-step action generation and Attention Residuals for depth-wise residual aggregation.",
"Report grounded socket-insert results from experimental_log.md: best iMF-AttnRes reaches avg_reward 1513.56 and median_reward 1901.5 with about 15.122 ms inference time, compared with diffusion-style infer100 baselines at 861.53/338.291 ms and 1019.39/397.912 ms.",
"Report grounded sim_transfer result: best iMF-AttnRes reaches avg_reward 526.22 and 44/100 success-like episodes compared with native diffusion policy at avg_reward 319.2 and 29/100 success-like episodes.",
"Mention limitations: simulation-only, heterogeneous hardware, and sensitivity to horizon/execution settings."
],
"citation_hints": []
},
{
"subsection_title": "Abstract Grounding Checks",
"content_bullets": [
"Ensure every numeric claim in the abstract is copied from experimental_log.md Table 1, Table 2, or Table 3.",
"Do not cite papers in the abstract unless the final manuscript style requires it; preserve double-blind anonymity."
],
"citation_hints": []
}
]
},
{
"section_title": "1. Introduction",
"subsections": [
{
"subsection_title": "1.1 Motivation and Problem Setting",
"content_bullets": [
"Motivate fast closed-loop manipulation in RoboIMI socket-insert and sim_transfer environments using the latency and reward metrics in experimental_log.md.",
"Explain that iterative diffusion-style inference can require infer100 evaluations in the baselines, producing hundreds of milliseconds per inference on socket-insert runs.",
"Frame the research question: can a one-to-few-step average-flow action generator preserve or improve manipulation reward while reducing inference latency?"
],
"citation_hints": [
"Chi et al. (Diffusion Policy: Visuomotor Policy Learning via Action Diffusion)",
"research paper or technical report introducing 'Vision-Language-Action robot policies'"
]
},
{
"subsection_title": "1.2 Contributions",
"content_bullets": [
"List contributions: adapting improved Mean Flow to VLA action generation, integrating AttnRes into the policy transformer, and evaluating on two RoboIMI simulated manipulation tasks.",
"State that the paper provides speed-quality comparisons against diffusion-style VLA, Diffusion Policy native DiT/DDPM/ResNet, ACT, and SmolVLA baselines using the exact 100-rollout statistics from experimental_log.md.",
"Emphasize honest scope: first draft simulation evidence rather than real-robot transfer."
],
"citation_hints": [
"Geng et al. (Mean Flows for One-step Generative Modeling)",
"Geng et al. (Improved Mean Flows: On the Challenges of Fast-Forward Generative Models)",
"research paper or technical report introducing 'Attention Residuals'"
]
}
]
},
{
"section_title": "2. Related Work",
"subsections": [
{
"subsection_title": "2.1 Diffusion and Flow Policies for Robot Action Generation",
"content_bullets": [
"Discuss diffusion-policy-style visuomotor action generation and its reliance on iterative denoising or repeated policy evaluations.",
"Connect diffusion and flow matching as continuous generative modeling tools, then distinguish iMF-AttnRes by its average-flow one-to-few-step inference objective.",
"Use related-work cluster 2.1 from outline.json to bridge to the method."
],
"citation_hints": [
"Chi et al. (Diffusion Policy: Visuomotor Policy Learning via Action Diffusion)",
"Ho et al. (Denoising Diffusion Probabilistic Models)",
"Lipman et al. (Flow Matching for Generative Modeling)",
"Liu et al. (Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow)",
"Peebles and Xie (Scalable Diffusion Models with Transformers)"
]
},
{
"subsection_title": "2.2 Vision-Language-Action Models and Action Chunking",
"content_bullets": [
"Summarize transformer and VLA robot policies as the broader setting for the proposed policy.",
"Discuss action chunking and compact VLA policies as complementary routes to practical control speed.",
"Position iMF-AttnRes as an action-generation module that can coexist with VLA conditioning and chunked execution."
],
"citation_hints": [
"Brohan et al. (RT-1: Robotics Transformer for Real-World Control at Scale)",
"Brohan et al. (RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control)",
"Kim et al. (OpenVLA: An Open-Source Vision-Language-Action Model)",
"Zhao et al. (Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware)",
"research paper or technical report introducing 'SmolVLA'"
]
},
{
"subsection_title": "2.3 Mean-Flow Training for One-Step Generation",
"content_bullets": [
"Use the derivations in idea.md to explain the distinction between instantaneous velocity v(x_t,t) and average velocity over [r,t].",
"Explain the Mean Flow relation between average velocity and instantaneous velocity and why the JVP term appears.",
"Explain the improved Mean Flow correction from idea.md: using the corrected relation to retain flow-matching training stability rather than directly substituting conditional velocity inside a nonlinear JVP expression."
],
"citation_hints": [
"Geng et al. (Mean Flows for One-step Generative Modeling)",
"Geng et al. (Improved Mean Flows: On the Challenges of Fast-Forward Generative Models)",
"Lipman et al. (Flow Matching for Generative Modeling)",
"Liu et al. (Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow)"
]
},
{
"subsection_title": "2.4 Residual Aggregation and Attention Residuals",
"content_bullets": [
"Derive the residual accumulation view in idea.md: x_t = y_0 + ... + y_t and y_{t+1}=f_{t+1}(sum y_s).",
"Explain the generalization to weighted residual aggregation and the AttnRes form a_{t+1,s} proportional to exp(w_{t+1} dot RMSNorm(y_s)).",
"Discuss why learned residual aggregation may help deeper policy networks avoid feature dilution and gradient imbalance."
],
"citation_hints": [
"He et al. (Deep Residual Learning for Image Recognition)",
"Vaswani et al. (Attention Is All You Need)",
"research paper or technical report introducing 'Attention Residuals'",
"research paper or technical report introducing 'PreNorm Transformers'"
]
}
]
},
{
"section_title": "3. Method",
"subsections": [
{
"subsection_title": "3.1 Problem Formulation",
"content_bullets": [
"Define the imitation-learning policy input as multi-view visual observations, robot state, and optional language/task conditioning from the RoboIMI environment, and define the output as an action chunk of length exec.",
"Introduce the evaluation setting from experimental_log.md: socket-insert and sim_transfer 100-rollout evaluations with reward, success-like count, inference FPS, control FPS, and inference time.",
"Avoid inventing dataset sizes, optimizer settings, or training details not present in idea.md or experimental_log.md."
],
"citation_hints": [
"research paper or technical report introducing 'behavior cloning for robotics'",
"research paper or technical report introducing 'Transformer architecture'"
]
},
{
"subsection_title": "3.2 Improved Mean Flow Action Generation",
"content_bullets": [
"Formalize the ODE dx_t/dt = v(x_t,t) and average velocity ̄v(z_t,r,t)=1/(t-r) int_r^t v(z_tau,tau)d tau using the notation from idea.md.",
"Show the Mean Flow identity ̄v = v - (t-r) d̄v/dt and the JVP expression jvp(̄v,(z,r,t),(v,0,1)).",
"Explain the iMF training relation V_theta(z_t)=u_theta(z_t)+(t-r)JVP_sg(u_theta;v_theta) and that gradients update the average-flow branch u_theta, supporting one-to-few-step inference."
],
"citation_hints": [
"Geng et al. (Mean Flows for One-step Generative Modeling)",
"Geng et al. (Improved Mean Flows: On the Challenges of Fast-Forward Generative Models)",
"Lipman et al. (Flow Matching for Generative Modeling)"
]
},
{
"subsection_title": "3.3 Attention Residual Policy Transformer",
"content_bullets": [
"Use the residual decomposition from idea.md to show fixed residual accumulation as a uniform sum over layer outputs.",
"Define AttnRes weights a_{t+1,s} with a softmax over w_{t+1} dot RMSNorm(y_s), then define the layer input as a weighted sum of previous residual outputs.",
"State that the paper uses AttnRes primarily in the policy transformer in the strongest variants; the full-AttnRes vision ablation underperformed in sim_transfer."
],
"citation_hints": [
"He et al. (Deep Residual Learning for Image Recognition)",
"Vaswani et al. (Attention Is All You Need)",
"research paper or technical report introducing 'Attention Residuals'"
]
},
{
"subsection_title": "3.4 Inference and Control Loop",
"content_bullets": [
"Describe one-to-few inference evaluations for iMF-AttnRes, using infer1/infer2/infer3 in socket runs and infer1 in sim_transfer best run.",
"Contrast this with diffusion-style infer100 baselines in experimental_log.md.",
"Explain that predicted chunks are executed for exec steps, e.g., exec16 in the best socket and sim_transfer iMF-AttnRes variants."
],
"citation_hints": [
"Chi et al. (Diffusion Policy: Visuomotor Policy Learning via Action Diffusion)",
"Zhao et al. (Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware)"
]
}
]
},
{
"section_title": "4. Experiments",
"subsections": [
{
"subsection_title": "4.1 Experimental Setup",
"content_bullets": [
"Describe RoboIMI socket-insert and sim_transfer environments as simulation benchmarks using only details stated in experimental_log.md.",
"Define metrics exactly as in experimental_log.md: avg_reward, median_reward, avg_max_reward, max_reward, nonzero_reward, success_like, avg_inference_fps, avg_control_fps, avg_inference_time_ms.",
"State that each row is based on 100 simulation rollouts and that hardware varies across run names, so speed is practical but not fully hardware-normalized."
],
"citation_hints": [
"research paper or technical report introducing 'Diffusion Policy: Visuomotor Policy Learning via Action Diffusion'",
"research paper or technical report introducing 'ACT action chunking transformer'",
"research paper or technical report introducing 'SmolVLA'"
]
},
{
"subsection_title": "4.2 Socket Peg Insertion Results",
"content_bullets": [
"Create a LaTeX table from Table 1 in experimental_log.md with the seven socket policies and their reward/speed metrics.",
"State that best iMF-AttnRes infer3 obtains avg_reward 1513.56 and median_reward 1901.5, exceeding diffusion-style ph16 infer100 avg_reward 861.53 and ph32 infer100 avg_reward 1019.39.",
"State that iMF-AttnRes infer2 latency is 14.409 ms and infer3 latency is 15.122 ms, compared with 338.291 ms and 397.912 ms for diffusion-style infer100 baselines.",
"Discuss the nonzero_reward caveat: diffusion-style ph16 has 97/100 nonzero episodes, but much lower median and average reward than iMF-AttnRes."
],
"citation_hints": []
},
{
"subsection_title": "4.3 Sim Transfer Results and Ablations",
"content_bullets": [
"Create a LaTeX table from Table 2 in experimental_log.md summarizing native diffusion policy, diffusion-style VLA, and iMF ablations.",
"Report that sim-transfer iMF-AttnRes infer1 reaches avg_reward 526.22 and 44/100 success-like episodes, compared with native diffusion policy at avg_reward 319.2 and 29/100 success-like episodes.",
"Analyze ablations: multi-token ResNet18 iMF is very fast at 441.80 FPS but lower reward; full-AttnRes vision underperforms; horizon/execution choices are sensitive."
],
"citation_hints": []
},
{
"subsection_title": "4.4 Speed-Quality Tradeoff and Failure Modes",
"content_bullets": [
"Use Table 3 from experimental_log.md for derived ratios: 1.76x socket reward over ph16 diffusion baseline, 22.97x socket inference FPS over ph16 diffusion baseline, and 1.65x sim_transfer average reward over native diffusion policy.",
"Discuss failure modes from experimental_log.md: simulation-only evidence, heterogeneous hardware, threshold differences across tasks, and sensitivity to horizon/execution and vision-residual design.",
"Emphasize that iMF-AttnRes improves latency but still needs controlled hardware-normalized studies and real-robot transfer validation."
],
"citation_hints": []
}
]
},
{
"section_title": "5. Limitations",
"subsections": [
{
"subsection_title": "5.1 Scope and Evaluation Limits",
"content_bullets": [
"Explicitly satisfy CoRL requirement by describing simulation-only evaluation and lack of real-robot experiments.",
"State heterogeneous hardware makes speed comparisons imperfect and motivates future hardware-normalized benchmarking.",
"State that iMF-AttnRes sensitivity to horizon/execution choices and full-AttnRes vision underperformance are current limitations."
],
"citation_hints": []
},
{
"subsection_title": "5.2 Future Work",
"content_bullets": [
"Propose future work: real-robot transfer, more seeds, hardware-normalized latency tests, and better criteria for where to apply AttnRes in vision vs policy modules.",
"Mention possible integration with efficient VLA backbones and action tokenization but avoid claiming untested results."
],
"citation_hints": []
}
]
},
{
"section_title": "6. Conclusion",
"subsections": [
{
"subsection_title": "Conclusion Content",
"content_bullets": [
"Summarize that improved Mean Flow plus AttnRes is a practical recipe for low-latency VLA action generation in the reported RoboIMI simulations.",
"Restate the strongest socket and sim_transfer quantitative evidence without adding new numbers.",
"Close with a balanced statement that this is promising simulation evidence requiring real-robot validation."
],
"citation_hints": []
},
{
"subsection_title": "Concluding Caveats",
"content_bullets": [
"End with simulation-only and hardware-normalization caveats from experimental_log.md rather than overclaiming real-robot deployment.",
"Mention future validation directions without adding unsupported experiments."
],
"citation_hints": []
}
]
}
]
}
+83
View File
@@ -0,0 +1,83 @@
{
"generated_at": "2026-05-15",
"note": "First PaperOrchestra draft. Figure rendering intentionally skipped per user request; Paper Banana prompts stored as placeholders in figures/*.paperbanana_prompt.txt and embedded in LaTeX tcolorboxes. Exa discovery was attempted; Semantic Scholar API key was not visible to shell, and unauthenticated S2 was rate-limited, so refs.bib was assembled from verified known paper metadata plus web/arXiv checks where possible.",
"files": [
{
"path": "inputs/idea.md",
"sha256": "537c18e73cf8d2e3f8c9ca52ca5570476659eba3e30ec53ec910eb93078bcf77",
"bytes": 13033
},
{
"path": "inputs/experimental_log.md",
"sha256": "4b08a21c2ac074583da43811f4dcea8fd17246830b584984e8d889e531e408e8",
"bytes": 11262
},
{
"path": "inputs/template.tex",
"sha256": "cd593743dfe04e9f0f954e20d0f06a66fc2c5848436bf99aca9b285b73f84724",
"bytes": 7653
},
{
"path": "inputs/conference_guidelines.md",
"sha256": "6580df5f779c3e71bbb8855ea68d53a9cce0ff99d678b4a7bf04b62ca29358e6",
"bytes": 12602
},
{
"path": "outline.json",
"sha256": "e8478a82156447db7ce5f5b87ce723bfdbc53bb5c831aad6c9be1756bf04a241",
"bytes": 25553
},
{
"path": "refs.bib",
"sha256": "80c6cd4eb09f996db097fc95f2309655a0264eb2d730bd451f992089e68654e8",
"bytes": 5693
},
{
"path": "citation_pool.json",
"sha256": "db92feda2d0ca640e16d02759f9bdb36c54efd86bb28ecb04b24cee6e3e3c7ef",
"bytes": 5371
},
{
"path": "figures/captions.json",
"sha256": "0666ac089fe9509fea18aaff1d3bfb7e61e2e4c06e53d1da91bc7d73e90b7f87",
"bytes": 2340
},
{
"path": "figures/fig_imf_attnres_components.paperbanana_prompt.txt",
"sha256": "8fc13b92afa4dc535145a372caeede673f34bd38b92411e005e5160b5bee804e",
"bytes": 551
},
{
"path": "figures/fig_method_overview.paperbanana_prompt.txt",
"sha256": "10d7219a1ad145a193dfeaef5a087986e1e08a2ebb93fe9bd64bd85e3e58c867",
"bytes": 579
},
{
"path": "figures/fig_sim_transfer_ablation.paperbanana_prompt.txt",
"sha256": "73edabb16bc28cc02d32722198f6f52eea2c5f54905b0171c738aaf7ac7d42e8",
"bytes": 505
},
{
"path": "figures/fig_socket_reward_latency.paperbanana_prompt.txt",
"sha256": "f535adb13a55cad29d1b03651f3fbc928b663c7c0e16d8a24880218123fc0da5",
"bytes": 571
},
{
"path": "final/paper.tex",
"sha256": "44964d95fa24183dc7e16f81b50b05fb7351e455e2d4771f1dc13791245c08e3",
"bytes": 28181
},
{
"path": "final/paper.pdf",
"sha256": "bf674c39629b089f6edf63253cdbf48e5fd2ce406cb9491807512032f4a28142",
"bytes": 240719
},
{
"path": "drafts/paper.tex",
"sha256": "44964d95fa24183dc7e16f81b50b05fb7351e455e2d4771f1dc13791245c08e3",
"bytes": 28181
}
],
"updated_at": "2026-05-16",
"last_change": "Moved mean-flow one-step generation derivation out of Related Work and into Method / Improved Mean Flow action generation."
}
+904
View File
@@ -0,0 +1,904 @@
{
"candidates": [
{
"title": "Visuomotor policy learning via action diffusion - ACM Digital Library",
"snippet": "Diffusion policy: : Visuomotor policy learning via action diffusion: International Journal of Robotics Research: Vol 4\n[...]\n, No 1\n[...]\nother-periodical;requested\n[...]\n:string:\n[...]\nPublication Websites;subPage:\n[...]\n:Basic Abstract\n[...]\n;page:string:Article/Chapter View;ctype:string:Journal Content;group\n[...]\n:acm-\n[...]\ntype>other-periodical;website:website:\n[...]\n-site;\n[...]\n:issue:\n[...]\n\\:10.5\n[...]\n55/rbrs.20\n[...]\n.44.issue-10-11;csubtype:string:Periodical;taxonomy:taxonomy:acm-pubtype;pageGroup:string:Publication Pages\"> skip to main content\n\n \n\n \n\nContents\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion mode",
"source_url": "https://dl.acm.org/doi/10.1177/02783649241273668",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://dl.acm.org/doi/10.1177/02783649241273668",
"_exa_published_date": null
},
{
"title": "Visuomotor Policy Learning via Action Diffusion - arXiv",
"snippet": "# Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 15 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details is available diffusion-policy.cs.columbia.edu\n[...]\nIn this work, we s",
"source_url": "https://arxiv.org/abs/2303.04137",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2303.04137",
"_exa_published_date": "2023-03-07T00:00:00.000Z"
},
{
"title": "[2303.04137v5] Diffusion Policy: Visuomotor Policy Learning via Action Diffusion",
"snippet": "[2303.04137v5] Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\n# Title:Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\n> Abstract:This paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robot's visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models.",
"source_url": "http://arxiv.org/abs/2303.04137v5",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "http://arxiv.org/abs/2303.04137v5",
"_exa_published_date": null
},
{
"title": "Visuomotor Policy Learning via Action Diffusion - arXiv",
"snippet": "Diffusion Policy\n\n# Diffusion Policy\n\nCheng Chi1, Siyuan Feng2, Yilun Du3, Zhenjia Xu1, Eric Cousineau2, Benjamin Burchfiel2, Shuran Song1 1 Columbia University 2 Toyota Research Institute 3 MIT https://diffusion-policy.cs.columbia.edu\n\n# Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n\nCheng Chi1, Siyuan Feng2, Yilun Du3, Zhenjia Xu1, Eric Cousineau2, Benjamin Burchfiel2, Shuran Song1 1 Columbia University 2 Toyota Research Institute 3 MIT https://diffusion-policy.cs.columbia.edu\n\n###### Abstract\n\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper",
"source_url": "https://arxiv.org/html/2502.12371v1",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/html/2502.12371v1",
"_exa_published_date": "2025-02-17T00:00:00.000Z"
},
{
"title": "[2303.04137] Diffusion Policy - ar5iv - arXiv",
"snippet": "# Diffusion Policy: Visuomotor Policy Learning via Action Diffusion\n[...]\nThis paper introduces Diffusion Policy, a new way of generating robot behavior by representing a robots visuomotor policy as a conditional denoising diffusion process. We benchmark Diffusion Policy across 12 different tasks from 4 different robot manipulation benchmarks and find that it consistently outperforms existing state-of-the-art robot learning methods with an average improvement of 46.9%. Diffusion Policy learns the gradient of the action-distribution score function and iteratively optimizes with respect to this gradient field during inference via a series of stochastic Langevin dynamics steps. We find that the diffusion formulation yields powerful advantages when used for robot policies, including gracefully handling multimodal action distributions, being suitable for high-dimensional action spaces, and exhibiting impressive training stability. To fully unlock the potential of diffusion models for visuomotor policy learning on physical robots, this paper presents a set of key technical contributions including the incorporation of receding horizon control, visual conditioning, and the time-series diffusion transformer. We hope this work will help motivate a new generation of policy learning techniques that are able to leverage the powerful generative modeling capabilities of diffusion models. Code, data, and training details will be publicly available.\n[...]\nIn this work, we seek to address thi",
"source_url": "https://ar5iv.labs.arxiv.org/html/2303.04137",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://ar5iv.labs.arxiv.org/html/2303.04137",
"_exa_published_date": null
},
{
"title": "[2006.11239v2] Denoising Diffusion Probabilistic Models",
"snippet": "[2006.11239\n[...]\n] Denoising Diffusion Probabilistic Models\n[...]\n# Title:Denoising Diffusion Probabilistic Models\n[...]\nAuthors: Jonathan Ho, Ajay Jain, Pieter Abbeel\n[...]\n> Abstract:We present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our implementation is available at this https URL\n\n \n\nhttps://doi.org/10.48550/arXiv.2006.11239\n\n \n\narXiv-issued DOI via DataCite",
"source_url": "https://arxiv.org/abs/2006.11239v2",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2006.11239v2",
"_exa_published_date": null
},
{
"title": "[2006.11239] Denoising Diffusion Probabilistic Models - arXiv",
"snippet": "[2006.11239] Denoising Diffusion Probabilistic Models\n[...]\n# Denoising Diffusion Probabilistic Models\n[...]\nJonathan Ho UC Berkeley jonathanho@berkeley.edu &Ajay Jain UC Berkeley ajayj@berkeley.edu &Pieter Abbeel UC Berkeley pabbeel@cs.berkeley.edu\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our implementation is available at https://github.com/hojonathanho/diffusion.\n[...]\nThis paper presents progress in diffusion probabilistic models [53]. A diffusion probabilistic model (which we will call a “diffusion model” for brevity) is a parameterized Markov chain trained using variational inference to produce samples matching the data after finite time. Transitions of this chain are learned to reverse a diffusion process, which is a Markov chain that gradually adds noise to the data in the opposite direction of s",
"source_url": "https://arxiv.org/abs/2006.11239",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2006.11239",
"_exa_published_date": "2020-06-19T00:00:00.000Z"
},
{
"title": "Denoising Diffusion Probabilistic Models",
"snippet": "Denoising Diffusion Probabilistic Models \n\nAuthorFeedback Bibtex MetaReview Paper Review Supplemental\n\n## Abstract\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models, a class of latent variable models inspired by considerations from nonequilibrium thermodynamics. Our best results are obtained by training on a weighted variational bound designed according to a novel connection between diffusion probabilistic models and denoising score matching with Langevin dynamics, and our models naturally admit a progressive lossy decompression scheme that can be interpreted as a generalization of autoregressive decoding. On the unconditional CIFAR10 dataset, we obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On 256x256 LSUN, we obtain sample quality similar to ProgressiveGAN.\n\n \n\nDo not remove: This comment is monitored to verify that the site is working properly",
"source_url": "https://papers.nips.cc/paper/2020/hash/4c5bcfec8584af0d967f1ab10179ca4b-Abstract.html",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://papers.nips.cc/paper/2020/hash/4c5bcfec8584af0d967f1ab10179ca4b-Abstract.html",
"_exa_published_date": null
},
{
"title": "",
"snippet": "Denoising Diffusion Probabilistic Models\n[...]\njonathanho@berkeley.edu\n[...]\nAjay Jain\n[...]\najayj@berkeley.edu\n[...]\nPieter Abbeel\n[...]\npabbeel@cs.berkeley.edu\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models,\n[...]\na class of latent variable models inspired by considerations from nonequilibrium\n[...]\nthermodynamics. Our best results are obtained by training on a weighted variational\n[...]\nbound designed according to a novel connection between diffusion probabilistic\n[...]\nmodels and denoising score matching with Langevin dynamics, and our models nat\u0002urally admit a progressive lossy decompression scheme that can be interpreted as a\n[...]\ngeneralization of autoregressive decoding. On the unconditional CIFAR10 dataset,\n[...]\nwe obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On\n[...]\n256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our imple\u0002mentation is available at https://github.com/hojonathanho/diffusion.\n[...]\nThis paper presents progress in diffusion probabilistic models [50]. A diffusion probabilistic model\n[...]\n(which we will call a “diffusion model” for brevity) is a parameterized Markov chain trained using\n[...]\nvariational inference to produce samples matching the data after finite time. Transitions of this chain\n[...]\nare learned to reverse a diffusion process, which is a Markov chain that gradually adds noise to the\n[...]\ndata in the opposite direction of sampling until signal",
"source_url": "https://arxiv.org/pdf/2006.11239v1",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/pdf/2006.11239v1",
"_exa_published_date": null
},
{
"title": "[PDF] Denoising Diffusion Probabilistic Models - arXiv",
"snippet": "Denoising Diffusion Probabilistic Models\n[...]\njonathanho@berkeley.edu\n[...]\nAjay Jain\n[...]\najayj@berkeley.edu\n[...]\nPieter Abbeel\n[...]\npabbeel@cs.berkeley.edu\n[...]\nWe present high quality image synthesis results using diffusion probabilistic models,\n[...]\na class of latent variable models inspired by considerations from nonequilibrium\n[...]\nthermodynamics. Our best results are obtained by training on a weighted variational\n[...]\nbound designed according to a novel connection between diffusion probabilistic\n[...]\nmodels and denoising score matching with Langevin dynamics, and our models nat\u0002urally admit a progressive lossy decompression scheme that can be interpreted as a\n[...]\ngeneralization of autoregressive decoding. On the unconditional CIFAR10 dataset,\n[...]\nwe obtain an Inception score of 9.46 and a state-of-the-art FID score of 3.17. On\n[...]\n256x256 LSUN, we obtain sample quality similar to ProgressiveGAN. Our imple\u0002mentation is available at https://github.com/hojonathanho/diffusion.\n[...]\nThis paper presents progress in diffusion probabilistic models [53]. A diffusion probabilistic model\n[...]\n(which we will call a “diffusion model” for brevity) is a parameterized Markov chain trained using\n[...]\nvariational inference to produce samples matching the data after finite time. Transitions of this chain\n[...]\nare learned to reverse a diffusion process, which is a Markov chain that gradually adds noise to the\n[...]\ndata in the opposite direction of sampling until signal",
"source_url": "https://arxiv.org/pdf/2006.11239",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/pdf/2006.11239",
"_exa_published_date": "2020-12-16T00:00:00.000Z"
},
{
"title": "[2212.09748] Scalable Diffusion Models with Transformers - arXiv",
"snippet": "William Peebles* UC Berkeley Saining Xie New York University\n[...]\nWe explore a new class of diffusion models based on the transformer architecture. We train latent diffusion models of images, replacing the commonly-used U-Net backbone with a transformer that operates on latent patches. We analyze the scalability of our Diffusion Transformers (DiTs) through the lens of forward pass complexity as measured by Gflops. We find that DiTs with higher Gflops—through increased transformer depth/width or increased number of input tokens—consistently have lower FID. In addition to possessing good scalability properties, our largest DiT-XL/2 models outperform all prior diffusion models on the class-conditional ImageNet 512 $\\times$ 512 and 256 $\\times$ 256 benchmarks, achieving a state-of-the-art FID of 2.27 on the latter.\n[...]\n, or Di\n[...]\nfor short. Di\n[...]\nadhere to the best practices of Vision Transformers (ViTs) [10], which have been shown to scale more effectively for visual recognition than\n[...]\nconvolutional networks (e.g., Res\n[...]\n[15]).\n[...]\nMore specifically, we study the scaling behavior of transformers with respect to network complexity vs. sample quality. We show that by constructing and benchmarking the DiT design space under the Latent Diffusion Models (LDMs) [48] framework, where diffusion models are trained within a VAEs latent space, we can successfully replace the U-Net backbone with a transformer. We further show that DiTs are scalable architectures for diff",
"source_url": "https://arxiv.org/abs/2212.09748",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://arxiv.org/abs/2212.09748",
"_exa_published_date": "2022-12-19T00:00:00.000Z"
},
{
"title": "Scalable Diffusion Models with Transformers | IEEE Conference Publication | IEEE Xplore",
"snippet": "Scalable Diffusion Models with Transformers | IEEE Conference Publication | IEEE Xplore\n\n \n\n \n\n \n\n### IEEE Account",
"source_url": "https://ieeexplore.ieee.org/document/10377858/",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://ieeexplore.ieee.org/document/10377858/",
"_exa_published_date": "2025-05-14T00:00:00.000Z"
},
{
"title": "[2212.09748v1] Scalable Diffusion Models with Transformers",
"snippet": "[2212.09748v1] Scalable Diffusion Models with Transformers\n[...]\n# Title:Scalable Diffusion Models with Transformers\n[...]\nAuthors: William Peebles, Saining Xie\n[...]\n> Abstract:We explore a new class of diffusion models based on the transformer architecture. We train latent diffusion models of images, replacing the commonly-used U-Net backbone with a transformer that operates on latent patches. We analyze the scalability of our Diffusion Transformers (DiTs) through the lens of forward pass complexity as measured by Gflops. We find that DiTs with higher Gflops -- through increased transformer depth/width or increased number of input tokens -- consistently have lower FID. In addition to possessing good scalability properties, our largest DiT-XL/2 models outperform all prior diffusion models on the class-conditional ImageNet 512x512 and 256x256 benchmarks, achieving a state-of-the-art FID of 2.27 on the latter.",
"source_url": "http://arxiv.org/abs/2212.09748v1",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "http://arxiv.org/abs/2212.09748v1",
"_exa_published_date": null
},
{
"title": "[PDF] Scalable Diffusion Models with Transformers | Semantic Scholar",
"snippet": "[PDF] Scalable Diffusion Models with Transformers | Semantic Scholar \n\nNavigate Paper Download (opens in a new tab) Share\n[...]\n```\n@article{Peebles2022ScalableDM,\n title={Scalable Diffusion Models with Transformers},\n author={William S. Peebles and Saining Xie},\n journal={2023 IEEE/CVF International Conference on Computer Vision (ICCV)},\n year={2022},\n pages={4172-4182},\n url={https://api.semanticscholar.org/CorpusID:254854389}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/736973165f98105fec3729b7db414ae4d80fcbeb",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://www.semanticscholar.org/reader/736973165f98105fec3729b7db414ae4d80fcbeb",
"_exa_published_date": "2022-12-19T14:39:50.000Z"
},
{
"title": "Scalable Diffusion Models with Transformers - IEEE Xplore",
"snippet": "Scalable Diffusion Models with Transformers | IEEE Conference Publication | IEEE Xplore\n**\n\n### IEEE Account\n* Change Username/Password\n* Update Address\n### Purchase Details\n* Payment Options\n* Order History\n* View Purchased Documents\n### Profile Information\n* Communications Preferences\n* Profession and Education\n* Technical Interests\n### Need Help?\n* **US & Canada:**+1 800 678 4333\n* **Worldwide:**+1 732 981 0060\n* Contact & Support\n* About IEEE*Xplore*\n* Contact Us\n* Help\n* Accessibility\n* Terms of Use\n* Nondiscrimination Policy\n* Sitemap\n* Privacy & Opting Out of Cookies\nA not-for-profit organization, IEEE is the world's largest technical professional organization dedicated to advancing technology for the benefit of humanity.\n© Copyright 2025 IEEE - All rights reserved. Use of this web site signifies your agreement to the terms and conditions.\n**",
"source_url": "https://ieeexplore.ieee.org/iel7/10376473/10376477/10377858.pdf",
"discovered_for": [
"rw.diffusion_policy"
],
"_exa_id": "https://ieeexplore.ieee.org/iel7/10376473/10376477/10377858.pdf",
"_exa_published_date": null
},
{
"title": "[PDF] Flow Matching for Generative Modeling - arXiv",
"snippet": "FLOW MATCHING FOR GENERATIVE MODELING\n[...]\nYaron Lipman1,2 Ricky T. Q. Chen1 Heli Ben-Hamu2 Maximilian Nickel1 Matt Le1\n[...]\nWe introduce a new paradigm for generative modeling built on Continuous\n[...]\n(CNFs), allowing us to train CNFs at unprecedented scale.\n[...]\nSpecifically, we present the notion of Flow Matching (FM), a simulation-free\n[...]\napproach for training CNFs based on regressing vector fields of fixed conditional\n[...]\nprobability paths. Flow Matching is compatible with a general family of Gaussian\n[...]\nprobability paths for transforming between noise and data samples—which\n[...]\nsubsumes existing diffusion paths as specific instances. Interestingly, we find\n[...]\nthat employing FM with diffusion paths results in a more robust and stable\n[...]\nalternative for training diffusion models. Furthermore, Flow Matching opens\n[...]\nthe door to training CNFs with other, non-diffusion probability paths. An\n[...]\ninstance of particular interest is using Optimal Transport (OT) displacement\n[...]\ninterpolation to define the conditional probability paths. These paths are more\n[...]\nefficient than diffusion paths, provide faster training and sampling, and result in\n[...]\nbetter generalization. Training CNFs using Flow Matching on ImageNet leads\n[...]\nto consistently better performance than alternative diffusion-based methods in\n[...]\nterms of both likelihood and sample quality, and allows fast and reliable sample\n[...]\ngeneration using off-the-shelf numerical ODE solvers.\n",
"source_url": "https://arxiv.org/pdf/2210.02747",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/pdf/2210.02747",
"_exa_published_date": "2023-02-08T00:00:00.000Z"
},
{
"title": "",
"snippet": "FLOW MATCHING FOR GENERATIVE MODELING\n[...]\nYaron Lipman1,2 Ricky T. Q. Chen1 Heli Ben-Hamu2 Maximilian Nickel1 Matt Le1\n[...]\nWe introduce a new paradigm for generative modeling built on Continuous\n[...]\nNormalizing Flows (CNFs), allowing us to train CNFs at unprecedented scale.\n[...]\nSpecifically, we present the notion of Flow Matching (FM), a simulation-free\n[...]\napproach for training CNFs based on regressing vector fields of fixed conditional\n[...]\nprobability paths. Flow Matching is compatible with a general family of Gaussian\n[...]\nprobability paths for transforming between noise and data samples—which\n[...]\nsubsumes existing diffusion paths as specific instances. Interestingly, we find\n[...]\nthat employing FM with diffusion paths results in a more robust and stable\n[...]\nalternative for training diffusion models. Furthermore, Flow Matching opens\n[...]\nthe door to training CNFs with other, non-diffusion probability paths. An\n[...]\ninstance of particular interest is using Optimal Transport (OT) displacement\n[...]\ninterpolation to define the conditional probability paths. These paths are more\n[...]\nefficient than diffusion paths, provide faster training and sampling, and result in\n[...]\nbetter generalization. Training CNFs using Flow Matching on ImageNet leads\n[...]\nto consistently better performance than alternative diffusion-based methods in\n[...]\nterms of both likelihood and sample quality, and allows fast and reliable sample\n[...]\ngeneration using off-the-shelf numer",
"source_url": "https://openreview.net/pdf?id=PqvMRDCJT9t",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://openreview.net/pdf?id=PqvMRDCJT9t",
"_exa_published_date": null
},
{
"title": "[2210.02747] Flow Matching for Generative Modeling - arXiv",
"snippet": "[221\n[...]\n] Flow Matching for Generative Modeling\n[...]\n# Flow Matching for Generative Modeling\n[...]\nYaron Lipman1,2 Ricky T. Q. Chen1 Heli Ben-Hamu2 Maximilian Nickel1 Matt Le1 1Meta AI (FAIR) 2Weizmann Institute of Science\n[...]\nWe introduce a new paradigm for generative modeling built on Continuous Normalizing Flows (CNFs), allowing us to train CNFs at unprecedented scale. Specifically, we present the notion of Flow Matching (FM), a simulation-free approach for training CNFs based on regressing vector fields of fixed conditional probability paths. Flow Matching is compatible with a general family of Gaussian probability paths for transforming between noise and data samples—which subsumes existing diffusion paths as specific instances. Interestingly, we find that employing FM with diffusion paths results in a more robust and stable alternative for training diffusion models. Furthermore, Flow Matching opens the door to training CNFs with other, non-diffusion probability paths. An instance of particular interest is using Optimal Transport (OT) displacement interpolation to define the conditional probability paths. These paths are more efficient than diffusion paths, provide faster training and sampling, and result in better generalization. Training CNFs using Flow Matching on ImageNet leads to consistently better performance than alternative diffusion-based methods in terms of both likelihood and sample quality, and allows fast and reliable sample generation using off-the-s",
"source_url": "https://arxiv.org/abs/2210.02747",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2210.02747",
"_exa_published_date": "2022-10-06T00:00:00.000Z"
},
{
"title": "[PDF] Flow Matching for Generative Modeling | Semantic Scholar",
"snippet": "```\n@article{Lipman2022FlowMF,\n title={Flow Matching for Generative Modeling},\n author={Yaron Lipman and Ricky T. Q. Chen and Heli Ben-Hamu and Maximilian Nickel and Matt Le},\n journal={ArXiv},\n year={2022},\n volume={abs/2210.02747},\n url={https://api.semanticscholar.org/CorpusID:252734897}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/af68f10ab5078bfc519caae377c90ee6d9c504e9",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://www.semanticscholar.org/reader/af68f10ab5078bfc519caae377c90ee6d9c504e9",
"_exa_published_date": "2022-10-06T03:39:18.000Z"
},
{
"title": "Flow Matching for Generative Modeling - OpenReview",
"snippet": "Flow Matching for Generative Modeling | OpenReview\n\n## Flow Matching for Generative Modeling\n\nICLR 2023 notable top 25%Readers: Everyone\n\nKeywords: continuous normalizing flows, generative models\n\nAbstract: We introduce a new paradigm for generative modeling built on Continuous Normalizing Flows (CNFs), allowing us to train CNFs at unprecedented scale. Specifically, we present the notion of Flow Matching (FM), a simulation-free approach for training CNFs based on regressing vector fields of fixed conditional probability paths. Flow Matching is compatible with a general family of Gaussian probability paths for transforming between noise and data samples---which subsumes existing diffusion paths as specific instances. Interestingly, we find that employing FM with diffusion paths results in a more robust and stable alternative for training diffusion models. Furthermore, Flow Matching opens the door to training CNFs with other, non-diffusion probability paths. An instance of particular interest is using Optimal Transport (OT) displacement interpolation to define the conditional probability paths. These paths are more efficient than diffusion paths, provide faster training and sampling, and result in better generalization. Training CNFs using Flow Matching on ImageNet leads to consistently better performance than alternative diffusion-based methods in terms of both likelihood and sample quality, and allows fast and reliable sample generation using off-the-shelf numerical ODE solve",
"source_url": "https://openreview.net/forum?id=PqvMRDCJT9t",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://openreview.net/forum?id=PqvMRDCJT9t",
"_exa_published_date": "2022-09-29T08:12:00.000Z"
},
{
"title": "[2505.13447] Mean Flows for One-step Generative Modeling - arXiv",
"snippet": "# Mean Flows for One-step Generative Modeling\n[...]\nZhengyang Geng1 Mingyang Deng2 Xingjian Bai2 J. Zico Kolter1 Kaiming He2 1CMU 2MIT Work partly done when visiting MIT.\n[...]\nWe propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the MeanFlow model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256 $\\times$ 256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.\n[...]\nIn this work, we propose a principled and effective framework, termed MeanFlow, for one-step generation. The core idea is to introduce a new ground-truth field representing the average velocity, in contrast to the instantaneous velocity typically modeled in Flow Matching. Average velocity is defined as the ratio of displacement to a time interval, with displacement give",
"source_url": "https://arxiv.org/abs/2505.13447",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2505.13447",
"_exa_published_date": "2025-05-19T00:00:00.000Z"
},
{
"title": "Mean Flows for One-step Generative Modeling",
"snippet": "# Mean Flows for One-step Generative Modeling\n[...]\nZhengyang Geng1 Mingyang Deng2 Xingjian Bai2 J. Zico Kolter1 Kaiming He2 1CMU 2MIT Work partly done when visiting MIT.\n[...]\nWe propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the MeanFlow model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256 $\\times$ 256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.\n[...]\nIn this work, we propose a principled and effective framework, termed MeanFlow, for one-step generation. The core idea is to introduce a new ground-truth field representing the average velocity, in contrast to the instantaneous velocity typically modeled in Flow Matching. Average velocity is defined as the ratio of displacement to a time interval, with displacement give",
"source_url": "https://arxiv.org/html/2505.13447",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/html/2505.13447",
"_exa_published_date": null
},
{
"title": "",
"snippet": "Mean Flows for One-step Generative Modeling\n[...]\nZhengyang Geng1 Mingyang Deng2 Xingjian Bai2 J. Zico Kolter1 Kaiming He2\n[...]\nWe propose a principled and effective framework for one-step generative modeling.\n[...]\nWe introduce the notion of average velocity to characterize flow fields, in contrast to\n[...]\ninstantaneous velocity modeled by Flow Matching methods. A well-defined identity\n[...]\nbetween average and instantaneous velocities is derived and used to guide neural\n[...]\nnetwork training. Our method, termed the MeanFlow model, is self-contained and\n[...]\nrequires no pre-training, distillation, or curriculum learning. MeanFlow demon\u0002strates strong empirical performance: it achieves an FID of 3.43 with a single\n[...]\nfunction evaluation (1-NFE) on ImageNet 256×256 trained from scratch, signifi\u0002cantly outperforming previous state-of-the-art one-step diffusion/flow models. Our\n[...]\nstudy substantially narrows the gap between one-step diffusion/flow models and\n[...]\ntheir multi-step predecessors, and we hope it will motivate future research to revisit\n[...]\nIn this work, we propose a principled and effective framework, termed MeanFlow, for one-step\n[...]\ngeneration. The core idea is to introduce a new ground-truth field representing the average velocity,\n[...]\nin contrast to the instantaneous velocity typically modeled in Flow Matching. Average velocity is\n[...]\ndefined as the ratio of displacement to a time interval, with displacement given by the time integral of\n[...",
"source_url": "https://openreview.net/pdf/611127035c61b58c8523c81094b8109e0d8f84fc.pdf",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://openreview.net/pdf/611127035c61b58c8523c81094b8109e0d8f84fc.pdf",
"_exa_published_date": "2025-10-24T00:00:00.000Z"
},
{
"title": "Mean Flows for One-step Generative Modeling - OpenReview",
"snippet": "Mean Flows for One-step Generative Modeling | OpenReview\n\n## Mean Flows for One-step Generative Modeling\n\n### Zhengyang Geng, Mingyang Deng, Xingjian Bai, J Zico Kolter, Kaiming He\n\nNeurIPS 2025 oralEveryone Revisions BibTeX CC BY 4.0\n\nKeywords: Generative Models\n\nAbstract: We propose a principled and effective framework for one-step generative modeling. We introduce the notion of average velocity to characterize flow fields, in contrast to instantaneous velocity modeled by Flow Matching methods. A well-defined identity between average and instantaneous velocities is derived and used to guide neural network training. Our method, termed the \\textit{MeanFlow} model, is self-contained and requires no pre-training, distillation, or curriculum learning. MeanFlow demonstrates strong empirical performance: it achieves an FID of 3.43 with a single function evaluation (1-NFE) on ImageNet 256$\\times$256 trained from scratch, significantly outperforming previous state-of-the-art one-step diffusion/flow models. Our study substantially narrows the gap between one-step diffusion/flow models and their multi-step predecessors, and we hope it will motivate future research to revisit the foundations of these powerful models.\n\nPrimary Area: Deep learning (e.g., architectures, generative models, optimization for deep networks, foundation models, LLMs)\n\nSubmission Number: 754\n\nLoading",
"source_url": "https://openreview.net/forum?id=uWj4s7rMnR",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://openreview.net/forum?id=uWj4s7rMnR",
"_exa_published_date": "2025-10-29T14:53:17.000Z"
},
{
"title": "[PDF] Mean Flows for One-step Generative Modeling | Semantic Scholar",
"snippet": "[PDF] Mean Flows for One-step Generative Modeling | Semantic Scholar\n[...]\n```\n@article{Geng2025MeanFF,\n title={Mean Flows for One-step Generative Modeling},\n author={Zhengyang Geng and Mingyang Deng and Xingjian Bai and J. Zico Kolter and Kaiming He},\n journal={ArXiv},\n year={2025},\n volume={abs/2505.13447},\n url={https://api.semanticscholar.org/CorpusID:278769814}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/19df654b0d0f634a451564346a09af8bd348dac0",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://www.semanticscholar.org/reader/19df654b0d0f634a451564346a09af8bd348dac0",
"_exa_published_date": "2025-05-19T06:36:24.000Z"
},
{
"title": "Improved Mean Flows: On the Challenges of Fastforward Generative ...",
"snippet": "# Improved Mean Flows: On the Challenges of Fastforward Generative Models\n[...]\nZhengyang Geng1,2,3, Yiyang Lu4,2, Zongze Wu3 Eli Shechtman3 J. Zico Kolter1 Kaiming He2 1CMU 2MIT 3Adobe 4THU Equal contribution. Part of this work was done when Z. Geng was interning at Adobe and MIT, and when Y. Lu was interning at MIT.\n[...]\nMeanFlow (MF) has recently been established as a framework for one-step generative modeling. However, its “fastforward” nature introduces key challenges in both the training objective and the guidance mechanism. First, the original MFs training target depends not only on the underlying ground-truth fields but also on the network itself. To address this issue, we recast the objective as a loss on the instantaneous velocity $v$ , re-parameterized by a network that predicts the average velocity $u$ . Our reformulation yields a more standard regression problem and improves the training stability. Second, the original MF fixes the classifier-free guidance scale during training, which sacrifices flexibility. We tackle this issue by formulating guidance as explicit conditioning variables, thereby retaining flexibility at test time. The diverse conditions are processed through in-context conditioning, which reduces model size and benefits performance. Overall, our improved MeanFlow (iMF) method, trained entirely from scratch, achieves 1.72 FID with a single function evaluation (1-NFE) on ImageNet 256 $\\times$ 256. iMF substantially outperforms prior methods of t",
"source_url": "https://arxiv.org/abs/2512.02012",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2512.02012",
"_exa_published_date": "2025-12-01T00:00:00.000Z"
},
{
"title": "",
"snippet": "probability path. At the same time, the design space around this training objective is rapidly expand\u0002ing with emerging techniques. For example, Mean Flow (Geng et al., 2025a) replaces instantaneous\n[...]\nvelocities with average velocities to enable strong one-step generation, improved Mean Flow vari\u0002ants (Geng et al., 2025b) reformulate the objective and guidance mechanism for better stability and\n[...]\nflexibility,\n[...]\n” (Li\n[...]\npatch Transformers.\n[...]\ncurriculum sampling, a\n[...]\n-phase curriculum that\n[...]\n(Logit-Normal)\n[...]\na coverage-focused distribution (Uniform). This method outperforms static base\u0002lines, achieving a 1\n[...]\n% relative improvement in FID over standard Uniform sampling\n[...]\nrevisits a central but often under-specified design choice in flow-based generative model\u0002ing: the timestep sampling distribution\n[...]\n). We identify a performance paradox in standard training:\n[...]\nMotivated by this analysis, we propose a two-phase curriculum for timestep sampling. The cur\u0002riculum uses a Logit-Normal distribution in a structure-learning phase to exploit fast early progress,\n[...]\nthen switches to Uniform sampling in a refinement phase to restore coverage of\n[...]\n-10, this simple schedule improves the best FID from 3.85 (Uniform) to 3.22 (a 16.4%\n[...]\nZhengyang Geng, Mingyang Deng, Xingjian Bai, J Zico Kolter, and Kaiming He. Mean flows for\n[...]\none-step generative modeling. NeurIPS, 2025a.\n[...]\nZhengyang Geng, Yiyang Lu, Zongze Wu, Eli Shechtman, J.",
"source_url": "https://arxiv.org/pdf/2603.12517",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/pdf/2603.12517",
"_exa_published_date": null
},
{
"title": "",
"snippet": "Tianhong Li 1 Zhengyang Geng 2 Kaiming He 1\n[...]\nModern diffusion/flow-based models for image\n[...]\nmade encouraging progress on each aspect in\u0002dividually, paving the way toward one-step dif\u0002fusion/flow without latents. In this work, we\n[...]\n“pixel MeanFlow” (pMF). Our core guideline is\n[...]\nformulate the network output space and the\n[...]\nloss space separately. The network target is de\u0002signed to be on a presumed low-dimensional im\u0002age manifold (i.e., x-prediction), while the loss\n[...]\nis defined via MeanFlow in the velocity space.\n[...]\nWe introduce a simple transformation between\n[...]\nthe image manifold and the average velocity field.\n[...]\nIn experiments, pMF achieves strong results for\n[...]\none-step latent-free generation on ImageNet at\n[...]\n256×256 resolution (2.22 FID) and 512×512 res\u0002olution (2.48 FID), filling a key missing piece in\n[...]\nIn this work, we propose pixel MeanFlow (pMF) for one\u0002step latent-free image generation. pMF follows the im\u0002proved MeanFlow (iMF) (Geng et al., 2025b) that learns\n[...]\nthe average velocity field (namely, u) using a loss defined\n[...]\nin the space of instantaneous velocity (namely, v). On the\n[...]\nother hand, following JiT (Li & He, 2025), pMF directly\n[...]\nparameterizes a denoised-image-like quantity (namely, x\u0002prediction), which is expected to lie on a low-dimensional\n[...]\nmanifold. To accommodate both formulations, we intro\u0002duce a conversion that relates the fields v, u, and x. We\n[...]\nempirically show that this formula",
"source_url": "https://arxiv.org/pdf/2601.22158",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/pdf/2601.22158",
"_exa_published_date": null
},
{
"title": "",
"snippet": "In this work, we propose Discrete MeanFlow (DMF) training curriculum, a budget-friendly frame\u0002work designed to bridge the gap between standard flow models and the MeanFlow identity for fast\n[...]\n. Our approach\n[...]\na staged curriculum\n[...]\nZhengyang Geng, Mingyang Deng, Xingjian Bai, J. Zico Kolter, and Kaiming He. Mean flows for\n[...]\n, 20\n[...]\n. URL https\n[...]\narxiv.org/\n[...]\n/2505.13447.\n[...]\nZhengyang Geng, Yiyang Lu, Zongze Wu, Eli Shechtman, J. Zico Kolter, and Kaiming He. Im\u0002proved mean flows: On the challenges of fastforward generative models, 2025b. URL https:\n[...]\n//arxiv.org/abs/2512.02012.",
"source_url": "https://arxiv.org/pdf/2604.08837",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/pdf/2604.08837",
"_exa_published_date": null
},
{
"title": "[2511.19065] Understanding, Accelerating, and Improving MeanFlow Training",
"snippet": "Accelerating, and Improving MeanFlow Training\n[...]\n# Title:Understanding, Accelerating, and Improving MeanFlow Training\n\nAuthors: Jin-Young Kim, Hyojun Go, Lea Bogensperger, Julius Erbach, Nikolai Kalischek, Federico Tombari, Konrad Schindler, Dominik Narnhofer\n\nView PDF HTML (experimental)\n[...]\n> Abstract:MeanFlow promises high-quality generative modeling in few steps, by jointly learning instantaneous and average velocity fields. Yet, the underlying training dynamics remain unclear. We analyze the interaction between the two velocities and find: (i) well-established instantaneous velocity is a prerequisite for learning average velocity; (ii) learning of instantaneous velocity benefits from average velocity when the temporal gap is small, but degrades as the gap increases; and (iii) task-affinity analysis indicates that smooth learning of large-gap average velocities, essential for one-step generation, depends on the prior formation of accurate instantaneous and small-gap average velocities. Guided by these observations, we design an effective training scheme that accelerates the formation of instantaneous velocity, then shifts emphasis from short- to long-interval average velocity. Our enhanced MeanFlow training yields faster convergence and significantly better few-step generation: With the same DiT-XL backbone, our method reaches an impressive FID of 2.87 on 1-NFE ImageNet 256x256, compared to 3.43 for the conventional MeanFlow baseline. Alternatively, our method matche",
"source_url": "https://arxiv.org/abs/2511.19065",
"discovered_for": [
"rw.mean_flow"
],
"_exa_id": "https://arxiv.org/abs/2511.19065",
"_exa_published_date": null
},
{
"title": "RT-1: Robotics Transformer for Real-World Control at Scale",
"snippet": "By transferring knowledge from large, diverse, task-agnostic datasets, modern machine learning models can solve specific downstream tasks either zero-shot or with small task-specific datasets to a high level of performance. While this capability has been demonstrated in other fields such as computer vision, natural language processing or speech recognition, it remains to be shown in robotics, where the generalization capabilities of the models are particularly critical due to the difficulty of collecting real-world robotic data. We argue that one of the keys to the success of such general robotic models lies with open-ended task-agnostic training, combined with high-capacity architectures that can absorb all of the diverse, robotic data. In this paper, we present a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties. We verify our conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks. The projects website and videos can be found at robotics-transformer1.github.io\n[...]\nThe second challenge lies in the design of the model itself. Effective robotic multi-task learning requires a high capacity model, and Transformer (Vaswani et al., 2017) models excel in this regard, particularly when it is necessary to learn many tasks conditioned, as in our case, on language instruct",
"source_url": "https://arxiv.org/html/2212.06817",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2212.06817",
"_exa_published_date": null
},
{
"title": "RT-1: Robotics Transformer for Real-World Control at Scale - arXiv",
"snippet": "By transferring knowledge from large, diverse, task-agnostic datasets, modern machine learning models can solve specific downstream tasks either zero-shot or with small task-specific datasets to a high level of performance. While this capability has been demonstrated in other fields such as computer vision, natural language processing or speech recognition, it remains to be shown in robotics, where the generalization capabilities of the models are particularly critical due to the difficulty of collecting real-world robotic data. We argue that one of the keys to the success of such general robotic models lies with open-ended task-agnostic training, combined with high-capacity architectures that can absorb all of the diverse, robotic data. In this paper, we present a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties. We verify our conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks. The projects website and videos can be found at robotics-transformer1.github.io\n[...]\nThe second challenge lies in the design of the model itself. Effective robotic multi-task learning requires a high capacity model, and Transformer (Vaswani et al., 2017) models excel in this regard, particularly when it is necessary to learn many tasks conditioned, as in our case, on language instruct",
"source_url": "https://arxiv.org/abs/2212.06817",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2212.06817",
"_exa_published_date": "2022-12-13T00:00:00.000Z"
},
{
"title": "RT-1: Robotics Transformer for Real-World Control at Scale - arXiv",
"snippet": "By transferring knowledge from large, diverse, task-agnostic datasets, modern machine learning models can solve specific downstream tasks either zero-shot or with small task-specific datasets to a high level of performance. While this capability has been demonstrated in other fields such as computer vision, natural language processing or speech recognition, it remains to be shown in robotics, where the generalization capabilities of the models are particularly critical due to the difficulty of collecting real-world robotic data. We argue that one of the keys to the success of such general robotic models lies with open-ended task-agnostic training, combined with high-capacity architectures that can absorb all of the diverse, robotic data. In this paper, we present a model class, dubbed Robotics Transformer, that exhibits promising scalable model properties. We verify our conclusions in a study of different model classes and their ability to generalize as a function of the data size, model size, and data diversity based on a large-scale data collection on real robots performing real-world tasks. The projects website and videos can be found at robotics-transformer1.github.io\n[...]\nThe second challenge lies in the design of the model itself. Effective robotic multi-task learning requires a high capacity model, and Transformer (Vaswani et al., 2017) models excel in this regard, particularly when it is necessary to learn many tasks conditioned, as in our case, on language instruct",
"source_url": "https://arxiv.org/html/2212.06817v2",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2212.06817v2",
"_exa_published_date": null
},
{
"title": "Bringing the RT-1-X Foundation Model to a SCARA robot",
"snippet": "Traditional robotic systems require specific training data for each task, environment, and robot form. While recent advancements in machine learning have enabled models to generalize across new tasks and environments, the challenge of adapting these models to entirely new settings remains largely unexplored. This study addresses this by investigating the generalization capabilities of the RT-1-X robotic foundation model to a type of robot unseen during its training: a SCARA robot from UMI-RTX.\n[...]\nRecent breakthroughs in machine learning and artificial intelligence suggest that training on large, diverse datasets can lead to highly\n[...]\nwhich often exceed\n[...]\nperformance of models developed for\n[...]\ndatasets tailored to\n[...]\nAs a result,\n[...]\nfield has been exploring more\n[...]\ncan adapt to\n[...]\n. Recent advancements like transformer\n[...]\ns RT-1 [brohan_rt-1_2022], which demonstrate the potential for\n[...]\nGoogles RT-1 model is an impressive work, tested on a collection of real-world robotic experiences, where in different institutes a fleet of robots were performing 700 tasks [brohan_rt-1_2022]. The robots in the training set, such as the Franka, Kuka iiwa, UR5 and the EveryDay robot, can move their end-effector in a spherical working-space. None of the robots in the dataset is of the SCARA (Selective Compliance Assembly Robot Arm) type. With a SCARA robot the movement of z-axis is decoupled from the movement in the x-y plane, which gives a SCARA robot an kidney ",
"source_url": "https://arxiv.org/html/2409.03299v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2409.03299v1",
"_exa_published_date": null
},
{
"title": "",
"snippet": "generalization capabilities of\n[...]\nfoundation models have\n[...]\ns RT-1\n[...]\nGoogles RT-1 model is an impressive work, tested on a collection of real\u0002\n[...]\nrobotic experiences, where in\n[...]\ninstitutes a fleet of robots were\n[...]\nstudy is to see if generalization capabilities of the Googles RT-1 model can\n[...]\nextended to an unseen robot\n[...]\nof a complete different type.\n[...]\nThe RT-1 model [2] was presented as a joint effort between Robotics at Google,\n[...]\nEveryday Robots, and Google Research, at the end of 2022. The purpose of RT\u00021 is to investigate if it is possible to train a single, capable, multi-task model\n[...]\non data consisting of a wide variety of robotic tasks, and to find out if such\n[...]\na model brings the same benefits observed in other domains, namely zero-shot\n[...]\ngeneralization to new tasks, environments, and objects.\n[...]\nWith RT-1, the authors present a model architecture along with a signif\u0002icant training dataset that fulfills those requirements, as well as demonstrate\n[...]\nof this model [2\n[...]\nBecause initial experiments [17] showed that zero-shot generalization was not\n[...]\npossible to the unseen UMI robotic embodiment, the decision was made to fine\u0002tune the model.\n[...]\nevaluate the generalization potential of\n[...]\n-1-\n[...]\nThis study explored the generalization capabilities of the RT-1-X model, particu\u0002larly its ability to adapt to a robot type not seen before. A dataset of demonstra\u0002tions was collected using the UMI robot, and",
"source_url": "https://arxiv.org/pdf/2409.03299",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2409.03299",
"_exa_published_date": "2024-09-06T00:00:00.000Z"
},
{
"title": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control | OpenReview",
"snippet": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control | OpenReview\n[...]\n## RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control\n[...]\nTL;DR: Vision-language models, trained on Internet-scale data, can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning.\n[...]\nAbstract: We study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows th",
"source_url": "https://openreview.net/forum?id=XMQgwiJ7KSX",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=XMQgwiJ7KSX",
"_exa_published_date": "2023-08-30T15:38:01.000Z"
},
{
"title": "[PDF] RT-2: Vision-Language-Action Models Transfer Web Knowledge to ...",
"snippet": "Transfer Web Knowledge\n[...]\n# RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control\n[...]\nWe study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows that our approach leads to performant robotic policies and enables RT-2 to obtain a range of emergent capabilities from Internet-scale training. This includes significantly improved generalization to novel objects, the ability to interpret commands not present in the robot t",
"source_url": "https://arxiv.org/pdf/2307.15818",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2307.15818",
"_exa_published_date": "2023-07-28T00:00:00.000Z"
},
{
"title": "[2307.15818] RT-2: Vision-Language-Action Models Transfer Web ...",
"snippet": "Transfer Web Knowledge\n[...]\n# RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control\n[...]\nWe study how vision-language models trained on Internet-scale data can be incorporated directly into end-to-end robotic control to boost generalization and enable emergent semantic reasoning. Our goal is to enable a single end-to-end trained model to both learn to map robot observations to actions and enjoy the benefits of large-scale pretraining on language and vision-language data from the web. To this end, we propose to co-fine-tune state-of-the-art vision-language models on both robotic trajectory data and Internet-scale vision-language tasks, such as visual question answering. In contrast to other approaches, we propose a simple, general recipe to achieve this goal: in order to fit both natural language responses and robotic actions into the same format, we express the actions as text tokens and incorporate them directly into the training set of the model in the same way as natural language tokens. We refer to such category of models as vision-language-action models (VLA) and instantiate an example of such a model, which we call RT-2. Our extensive evaluation (6k evaluation trials) shows that our approach leads to performant robotic policies and enables RT-2 to obtain a range of emergent capabilities from Internet-scale training. This includes significantly improved generalization to novel objects, the ability to interpret commands not present in the robot t",
"source_url": "https://arxiv.org/abs/2307.15818",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2307.15818",
"_exa_published_date": "2023-07-28T00:00:00.000Z"
},
{
"title": "",
"snippet": "language and point clouds and varying sequence lengths (from 200 to few thousand). Left: A 5B vision-language-action model from\n[...]\nthe RT-2 class [1] (sequence length: L = 196). The manipulation policy is conditioned on the text instruction. Right: A Point Cloud\n[...]\n-parameter vision-language-action models or\n[...]\nAs), into\n[...]\nspeeding up: (a) the class of recently introduced RT-2 models\n[...]\n[1], the first VLA robotic policies pre-trained on internet\u0002scale data, as well as (b) Point Cloud Transformer (PCT)\n[...]\n[12]), multi-modal sensor fusion [16], finally the first vision\u0002language-action robotic manipulation powered by massive\n[...]\nvision language models [1].\n[...]\nparameter models such as RT-2 [1].\n[...]\nto convert pre-trained or already fine-tuned Transformer\u0002based robotic policies of quadratic space and time complex\u0002ity (including massive billion-parameter vision-language\u0002action models or VLAs), into their efficient linear-attention\n[...]\neffectiveness of SARA-RT by speeding up: (a) the class\n[...]\nof the aforementioned RT-2 models, the first VLA robotic\n[...]\npolicies pre-trained on internet-scale data, as well as (b)\n[...]\nB. RT-2 vision-language-action models\n[...]\n1) The setting: Next we consider the class of RT-2 archi\u0002tectures from [1]. Those apply PaLI-X [36]) vision-language\u0002model (VLM) backbones to encode policies taking vision\n[...]\ninput and conditioned on natural language instructions. We\n[...]\nfocus on the 5B PaLI-X variant, as more practical ",
"source_url": "https://arxiv.org/pdf/2312.01990",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2312.01990",
"_exa_published_date": null
},
{
"title": "",
"snippet": "robotics applications is particularly\n[...]\n. One notable\n[...]\nexample is RT-2, a system capable of generating low-level actions\n[...]\nrepresented in textual format from a given instruction alongside\n[...]\na sequence of historical actions and image observations. To\n[...]\nstimulate further research in this domain, we introduce an\n[...]\nopen-source implementation tailored for utilizing VLMs in\n[...]\ninstruction-based robot control. This implementation supports\n[...]\na variety of VLM architectures and facilitates straightforward\n[...]\nintegration of new models. We use our framework to train\n[...]\nmultiple VLMs and evaluate them on a physical robot. The\n[...]\nresults validate the practical efficacy of our framework, thus\n[...]\npaving the way for enhanced understanding and capabilities in\n[...]\ninstruction-based robot control systems. The code is available\n[...]\nA notable endeavor in this domain is RT-2 [12], which\n[...]\nemploys a VLM to interpret an instruction, past actions\n[...]\nand image observations, and generate subsequent actions in\n[...]\ntextual form. RT-2 has demonstrated the potential of creating\n[...]\ninstruction-based low-level robot control policies, showcasing\n[...]\nremarkable performance and notable generalization capabil\u0002ities. However, RT-2s unavailability to the public and its\n[...]\nconsiderable scale, with 55 billion parameters, pose challenges\n[...]\nfor academic endeavors to engage with it effectively, hindering\n[...]\nfurther exploration and refinement of thi",
"source_url": "https://openreview.net/pdf?id=nfm2qcV1S4",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/pdf?id=nfm2qcV1S4",
"_exa_published_date": null
},
{
"title": "OpenVLA: An Open-Source Vision-Language-Action Model - arXiv",
"snippet": "OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\n# OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\nLarge policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challenges, we introduce OpenVLA, a 7B-parameter open-source VLA trained on a diverse collection of 970k real-world robot demonstrations. OpenVLA builds on a Llama 2 language model combined with a visual encoder that fuses pretrained features from DINOv2 and SigLIP. As a product of the added data diversity and new model components, OpenVLA demonstrates strong results for generalist manipulation, outperforming closed models such as RT-2-X (55B) by 16.5% in absolute task success rate across 29 tasks and multiple robot embodiments, with 7x fewer parameters. We further show that we can effectively fine-tune OpenVLA for new settings, with especially strong generalization results in multi-task environments involving multiple objects and strong language",
"source_url": "https://arxiv.org/abs/2406.09246",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2406.09246",
"_exa_published_date": "2024-06-13T00:00:00.000Z"
},
{
"title": "[PDF] OpenVLA: An Open-Source Vision-Language-Action Model | Semantic Scholar",
"snippet": "[PDF] OpenVLA: An Open-Source Vision-Language-Action Model | Semantic Scholar \n\nNavigate Paper Download (opens in a new tab) Share\n[...]\n```\n@article{Kim2024OpenVLAAO,\n title={OpenVLA: An Open-Source Vision-Language-Action Model},\n author={Moo Jin Kim and Karl Pertsch and Siddharth Karamcheti and Ted Xiao and Ashwin Balakrishna and Suraj Nair and Rafael Rafailov and Ethan Paul Foster and Grace Lam and Pannag R. Sanketi and Quan Vuong and Thomas Kollar and Benjamin Burchfiel and Russ Tedrake and Dorsa Sadigh and Sergey Levine and Percy Liang and Chelsea Finn},\n journal={ArXiv},\n year={2024},\n volume={abs/2406.09246},\n url={https://api.semanticscholar.org/CorpusID:270440391}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/8f9ceb5ffad8e7a066dfc9d9aaa5153b714740ee",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://www.semanticscholar.org/reader/8f9ceb5ffad8e7a066dfc9d9aaa5153b714740ee",
"_exa_published_date": "2024-06-13T09:02:12.000Z"
},
{
"title": "OpenVLA: An Open-Source Vision-Language-Action Model",
"snippet": "OpenVLA: An Open-Source Vision-Language-Action Model | OpenReview\n\n## OpenVLA: An Open-Source Vision-Language-Action Model\n\n### Moo Jin Kim, Karl Pertsch, Siddharth Karamcheti, Ted Xiao, Ashwin Balakrishna, Suraj Nair, Rafael Rafailov, Ethan P Foster, Pannag R Sanketi, Quan Vuong, Thomas Kollar, Benjamin Burchfiel, Russ Tedrake, Dorsa Sadigh, Sergey Levine, Percy Liang, Chelsea Finn \n\nCoRL 2024everyonesince 05 Sept 2024\">Everyone Revisions BibTeX CC BY 4.0\n\nKeywords: Vision-Language-Action Models, Generalist Policies, Large-scale Robot Learning, Robotic Manipulation, Robotics, Vision-Language Models\n\nTL;DR: We introduce OpenVLA, a state-of-the-art, open-source 7B-parameter VLA model that obtains strong performance for cross-embodiment robot control out-of-the-box and can be easily adapted to new robot setups via parameter-efficient fine-tuning.\n\nAbstract: Large policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challeng",
"source_url": "https://openreview.net/forum?id=ZMnD6QZAE6",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=ZMnD6QZAE6",
"_exa_published_date": "2024-09-05T00:00:00.000Z"
},
{
"title": "[2406.09246v3] OpenVLA: An Open-Source Vision-Language-Action Model",
"snippet": "[2406.09246v3] OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\n# Title:OpenVLA: An Open-Source Vision-Language-Action Model\n[...]\n> Abstract:Large policies pretrained on a combination of Internet-scale vision-language data and diverse robot demonstrations have the potential to change how we teach robots new skills: rather than training new behaviors from scratch, we can fine-tune such vision-language-action (VLA) models to obtain robust, generalizable policies for visuomotor control. Yet, widespread adoption of VLAs for robotics has been challenging as 1) existing VLAs are largely closed and inaccessible to the public, and 2) prior work fails to explore methods for efficiently fine-tuning VLAs for new tasks, a key component for adoption. Addressing these challenges, we introduce OpenVLA, a 7B-parameter open-source VLA trained on a diverse collection of 970k real-world robot demonstrations. OpenVLA builds on a Llama 2 language model combined with a visual encoder that fuses pretrained features from DINOv2 and SigLIP. As a product of the added data diversity and new model components, OpenVLA demonstrates strong results for generalist manipulation, outperforming closed models such as RT-2-X (55B) by 16.5% in absolute task success rate across 29 tasks and multiple robot embodiments, with 7x fewer parameters. We further show that we can effectively fine-tune OpenVLA for new settings, with especially strong generalization results in multi-task environments involving mult",
"source_url": "https://arxiv.org/abs/2406.09246v3",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2406.09246v3",
"_exa_published_date": null
},
{
"title": "Revision History for OpenVLA: An Open-Source... - OpenReview",
"snippet": "Revisions | OpenReview\n\nLoading",
"source_url": "https://openreview.net/revisions?id=ZMnD6QZAE6",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/revisions?id=ZMnD6QZAE6",
"_exa_published_date": null
},
{
"title": "A Vision-Language-Action Model for Affordable and Efficient Robotics",
"snippet": "SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\n# SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\nVision-language models (VLMs) pretrained on large-scale multimodal datasets encode rich visual and linguistic knowledge, making them a strong foundation for robotics. Rather than training robotic policies from scratch, recent approaches adapt VLMs into vision-language-action (VLA) models that enable natural language-driven perception and control. However, existing VLAs are typically massiveoften with billions of parametersleading to high training costs and limited real-world deployability. Moreover, they rely on academic and industrial datasets, overlooking the growing availability of community-collected data from affordable robotic platforms. In this work, we present SmolVLA, a small, efficient, and community-driven VLA that drastically reduces both training and inference costs, while retaining competitive performance. SmolVLA is designed to be trained on a single GPU and deployed on consumer-grade GPUs or even CPUs. To further improve responsiveness, we introduce an asynchronous inference stack decoupling perception and action prediction from action execution, allowing higher control rates with chunked action generation. Despite its compact size, SmolVLA achieves performance comparable to VLAs that are 10 $\\times$ larger. We evaluate SmolVLA on a range of both simulated as well as real-world robotic bench",
"source_url": "https://arxiv.org/abs/2506.01844",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2506.01844",
"_exa_published_date": "2025-06-02T00:00:00.000Z"
},
{
"title": "[PDF] SmolVLA: A Vision-Language-Action Model for Affordable ... - arXiv",
"snippet": "SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\n# SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\nVision-language models (VLMs) pretrained on large-scale multimodal datasets encode rich visual and linguistic knowledge, making them a strong foundation for robotics. Rather than training robotic policies from scratch, recent approaches adapt VLMs into vision-language-action (VLA) models that enable natural language-driven perception and control. However, existing VLAs are typically massiveoften with billions of parametersleading to high training costs and limited real-world deployability. Moreover, they rely on academic and industrial datasets, overlooking the growing availability of community-collected data from affordable robotic platforms. In this work, we present SmolVLA, a small, efficient, and community-driven VLA that drastically reduces both training and inference costs, while retaining competitive performance. SmolVLA is designed to be trained on a single GPU and deployed on consumer-grade GPUs or even CPUs. To further improve responsiveness, we introduce an asynchronous inference stack decoupling perception and action prediction from action execution, allowing higher control rates with chunked action generation. Despite its compact size, SmolVLA achieves performance comparable to VLAs that are 10 $\\times$ larger. We evaluate SmolVLA on a range of both simulated as well as real-world robotic bench",
"source_url": "https://arxiv.org/pdf/2506.01844",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2506.01844",
"_exa_published_date": "2025-06-02T00:00:00.000Z"
},
{
"title": "SmolVLA: A vision-language-action model for affordable and ... - arXiv",
"snippet": "SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\n# SmolVLA: A vision-language-action model for affordable and efficient robotics\n[...]\nVision-language models (VLMs) pretrained on large-scale multimodal datasets encode rich visual and linguistic knowledge, making them a strong foundation for robotics. Rather than training robotic policies from scratch, recent approaches adapt VLMs into vision-language-action (VLA) models that enable natural language-driven perception and control. However, existing VLAs are typically massiveoften with billions of parametersleading to high training costs and limited real-world deployability. Moreover, they rely on academic and industrial datasets, overlooking the growing availability of community-collected data from affordable robotic platforms. In this work, we present SmolVLA, a small, efficient, and community-driven VLA that drastically reduces both training and inference costs, while retaining competitive performance. SmolVLA is designed to be trained on a single GPU and deployed on consumer-grade GPUs or even CPUs. To further improve responsiveness, we introduce an asynchronous inference stack decoupling perception and action prediction from action execution, allowing higher control rates with chunked action generation. Despite its compact size, SmolVLA achieves performance comparable to VLAs that are 10 $\\times$ larger. We evaluate SmolVLA on a range of both simulated as well as real-world robotic bench",
"source_url": "https://arxiv.org/html/2506.01844v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2506.01844v1",
"_exa_published_date": "2025-06-02T00:00:00.000Z"
},
{
"title": "[PDF] SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics | Semantic Scholar",
"snippet": "[PDF] SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics | Semantic Scholar \n\nNavigate Paper Download (opens in a new tab) Share\n[...]\n```\n@article{Shukor2025SmolVLAAV,\n title={SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},\n author={Mustafa Shukor and Dana Aubakirova and Francesco Capuano and Pepijn Kooijmans and Steven Palma and Adil Zouitine and Michel Aractingi and Caroline Pascal and Martino Russi and Andr{\\'e}s Marafioti and Simon Alibert and Matthieu Cord and Thomas Wolf and R{\\'e}mi Cad{\\`e}ne},\n journal={ArXiv},\n year={2025},\n volume={abs/2506.01844},\n url={https://api.semanticscholar.org/CorpusID:279119427}\n}\n```",
"source_url": "https://www.semanticscholar.org/reader/6ab4d113676d00e74b55e918fee4c7affaa8652f",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://www.semanticscholar.org/reader/6ab4d113676d00e74b55e918fee4c7affaa8652f",
"_exa_published_date": "2025-06-02T13:54:14.000Z"
},
{
"title": "Lite VLA: Efficient Vision-Language-Action Control on CPU-Bound Edge Robots",
"snippet": "By leveraging NF4 quantization and the llama-cpp runtime, the proposed LiteVLA implementation pioneers the CPU-only deployment path, achieving functional asynchronous visuomotor control on the low-cost Raspberry Pi 4. This represents a novel deployment strategy not demonstrated by prior GPU-centric VLA frameworks such as SmolVLA by Shukor et al. [22], whose work focused primarily on static robotic arms. Beyond proving technical feasibility, this work establishes a scalable methodology for deploying generalist robot intelligence under strict computational budgets.\n[...]\nParameter-efficient adaptation. We fine-tune a compact SmolVLM backbone using LoRA (rank 8, $\\alpha{=}8$ , dropout 0.1) to specialize visuomotor translation under tight memory/compute budgets (Alg. 1; Sec. III, pp. 23).\n[...]\n. 4).\n[...]\nLarge-scale multimodal systems such as PaLM-E, SayCan, and RT-2 have shown that unified language-conditioned reasoning enables robots to follow natural language commands and execute complex manipulation tasks. However, these approaches rely heavily on cloud-based computation and high-end GPUs, making them impractical for resource-limited or field-deployed robots. SMolVLA by Shukor et al. [22] introduced a small and efficient vision-language-action framework designed for community-driven robotic experimentation. It demonstrated that compact multimodal transformers could achieve competitive visuomotor reasoning performance while running on consumer-grade GPUs or CPUs. Nonetheles",
"source_url": "https://arxiv.org/html/2511.05642",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2511.05642",
"_exa_published_date": null
},
{
"title": "A Vision-Language-Action Flow Model for General Robot Control",
"snippet": "𝜋₀: A Vision-Language-Action Flow Model for General Robot Control\n[...]\n# $\\pi_{0}$ : A Vision-Language-Action Flow Model for General Robot Control\n[...]\nholds tremendous promise to unlock\n[...]\nfull potential of flexible, general, and dexterous\n[...]\nsystems, as well as to address some of\n[...]\ndeepest questions in artificial intelligence\n[...]\nHowever, bringing robot learning to\n[...]\nlevel of generality required for effective real-world systems faces major obstacles in terms of data, generalization, and robustness. In this paper, we discuss how generalist robot policies (i.e., robot foundation models) can address these challenges, and how\n[...]\ncan design effective generalist robot policies for complex and highly dexterous tasks. We propose a novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge. We then discuss how this model can be trained on a large and diverse dataset from multiple dexterous robot platforms, including single-arm robots, dual-arm robots, and mobile manipulators. We evaluate our model in terms of\n[...]\nability to perform tasks via direct prompting, follow language instructions from people and from a high-level VLM policy, and\n[...]\nability to acquire new skills via fine-tuning. Our results cover a wide variety of tasks, such as laundry folding, table cleaning, and assembling boxes.\n[...]\nIn this paper, we present a prototype model and learning framework, which we call $\\pi_",
"source_url": "https://arxiv.org/abs/2410.24164",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2410.24164",
"_exa_published_date": "2024-10-31T00:00:00.000Z"
},
{
"title": "π: A Vision-Language-Action Flow Model for General Robot Control | OpenReview",
"snippet": "π: A Vision-Language-Action Flow Model for General Robot Control | OpenReview\n[...]\n## π: A Vision-Language-Action Flow Model for General Robot Control\n[...]\nAbstract: Robot learning holds tremendous promise to unlock the full potential of flexible, general, and dexterous robot systems, as well as to address some of the deepest questions in artificial intelligence. However, bringing robot learning to the level of generality required for effective real-world systems faces major obstacles in terms of data, generalization, and robustness. In this paper, we discuss how generalist robot policies (i.e., robot foundation models) can address these challenges, and how we can design effective generalist robot policies for complex and highly dexterous tasks. We propose a novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge. We then discuss how this model can be trained on a large and diverse dataset from multiple dexterous robot platforms, including single-arm robots, dual-arm robots, and mobile manipulators. We evaluate our model in terms of its ability to perform tasks in zero shot after pre-training, follow language instructions from people and from a high-level VLM policy, and its ability to acquire new skills via fine-tuning. Our results cover a wide variety of tasks, such as laundry folding, table cleaning, and assembling boxes.",
"source_url": "https://openreview.net/forum?id=38a45ho9Nq",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=38a45ho9Nq",
"_exa_published_date": null
},
{
"title": "π0: A Vision-Language-Action Flow Model for General Robot Control",
"snippet": "# π0: A Vision-Language-Action Flow Model for General Robot Control\n[...]\n```\n@article{Black20240AV,\n title={$\\pi$0: A Vision-Language-Action Flow Model for General Robot Control},\n author={Kevin Black and Noah Brown and Danny Driess and Adnan Esmail and Michael Equi and Chelsea Finn and Niccolo Fusai and Lachy Groom and Karol Hausman and Brian Ichter and Szymon Jakubczak and Tim Jones and Liyiming Ke and Sergey Levine and Adrian Li-Bell and Mohith Mothukuri and Suraj Nair and Karl Pertsch and Lucy Xiaoyang Shi and James Tanner and Quan Vuong and Anna Walling and Haohuan Wang and Ury Zhilinsky},\n journal={ArXiv},\n year={2024},\n volume={abs/2410.24164},\n url={https://api.semanticscholar.org/CorpusID:273811174}\n}\n\n```\n[...]\n- Kevin Black, Noah Brown, +21 authors Ury Zhilinsky\n[...]\n- Published in arXiv.org 31 October\n[...]\n2024\n[...]\n- Computer Science, Engineering\n[...]\nTLDR\n\nA novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge is proposed and evaluated in terms of its ability to perform tasks in zero shot after pre-training, follow language instructions from people and from a high-level VLM policy, and its ability to acquire new skills via fine-tuning.Expand",
"source_url": "https://www.semanticscholar.org/paper/%CF%800%3A-A-Vision-Language-Action-Flow-Model-for-General-Black-Brown/7e7e59d2e247d99954081080ddd5aae93d10b9e0",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://www.semanticscholar.org/paper/%CF%800%3A-A-Vision-Language-Action-Flow-Model-for-General-Black-Brown/7e7e59d2e247d99954081080ddd5aae93d10b9e0",
"_exa_published_date": null
},
{
"title": "𝜋₀: A Vision-Language-Action Flow Model for General Robot Control",
"snippet": "𝜋₀: A Vision-Language-Action Flow Model for General Robot Control\n[...]\n# $\\pi_{0}$ : A Vision-Language-Action Flow Model for General Robot Control\n[...]\nRobot learning holds tremendous promise to unlock the full potential of flexible, general, and dexterous robot systems, as well as to address some of the deepest questions in artificial intelligence\n[...]\nHowever, bringing robot learning to\n[...]\nlevel of generality required for effective real-world systems faces major obstacles in terms of data, generalization, and robustness. In this paper, we discuss how generalist robot policies (i.e., robot foundation models) can address these challenges, and how we can design effective generalist robot policies for complex and highly dexterous tasks. We propose a novel flow matching architecture built on top of a pre-trained vision-language model (VLM) to inherit Internet-scale semantic knowledge. We then discuss how this model can be trained on a large and diverse dataset from multiple dexterous robot platforms, including single-arm robots, dual-arm robots, and mobile manipulators. We evaluate our model in terms of its ability to perform tasks in zero shot after pre-training, follow language instructions from people and from a high-level VLM policy, and\n[...]\nability to acquire new skills via fine-tuning. Our results cover a wide variety of tasks, such as laundry folding, table cleaning, and assembling boxes.\n[...]\nIn this paper, we present a prototype model and learning framework, wh",
"source_url": "https://arxiv.org/html/2410.24164v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2410.24164v1",
"_exa_published_date": "2024-10-31T00:00:00.000Z"
},
{
"title": "",
"snippet": "We train three policy classes: MolmoBot, a Molmo2-based multi-frame vision-language model with\n[...]\na flow-matching action head; MolmoBot-Pi0, which replicates the π0 architecture to enable direct\n[...]\ncomparison; and MolmoBot-SPOC, a lightweight policy suitable for edge deployment and amenable to\n[...]\nNVIDIAs GR00T [1], Physical Intelligences π0 [2, 3], and Google DeepMinds Gemini Robotics [4] frames\n[...]\nlarge-scale real-world training as the basis for generalist manipulation agents that act in the physical world.\n[...]\nMolmoBot-Pi0\n[...]\nUsing this data, we train three policy classes. Our flagship model, MolmoBot, is built on top of Molmo2 [7], our\n[...]\nvideo-language model capable of ingesting past frames for context. We augment this architecture with a DiT\u0002based flow-matching action head that is layerwise coupled to the vision-language backbone. Each action layer\n[...]\ncross-attends to the corresponding intermediate hidden states of the underlying VLM, while also incorporating\n[...]\nrobot-state features, allowing actions to be generated from multi-scale multimodal representations.\n[...]\nfrom MolmoBot, we also train MolmoBot-Pi0, which exactly replicates the π0 architecture for controlled\n[...]\ncomparison; and MolmoBot-SPOC, a lightweight non-VLA policy suitable for edge deployment and future RL\n[...]\nfine-tuning.\n[...]\nWe provide ablations demonstrating\n[...]\nimportance of data scale and diversity, and show through MolmoBot\u0002Pi0 that our\n[...]\nyields strong perfor",
"source_url": "https://arxiv.org/pdf/2603.16861",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/pdf/2603.16861",
"_exa_published_date": null
},
{
"title": "Efficient Action Tokenization for Vision-Language-Action Models",
"snippet": "Autoregressive sequence models, such as Transformer-based vision-language action (VLA) policies, can be tremendously effective for capturing complex and generalizable robotic behaviors. However, such models require us to choose a tokenization of our continuous action signals, which determines how the discrete symbols predicted by the model map to continuous robot actions. We find that current approaches for robot action tokenization, based on simple per-dimension, per-timestep binning schemes, typically perform poorly when learning dexterous skills from high-frequency robot data. To address this challenge, we propose a new compression-based tokenization scheme for robot actions, based on the discrete cosine transform. Our tokenization approach, Frequency-space Action Sequence Tokenization (FAST), enables us to train autoregressive VLAs for highly dexterous and high-frequency tasks where standard discretization methods fail completely. Based on FAST, we release FAST+, a universal robot action tokenizer, trained on 1M real robot action trajectories. It can be used as a black-box tokenizer for a wide range of robot action sequences, with diverse action spaces and control frequencies. Finally, we show that, when combined with the $\\bm{\\pi_{0}}$ VLA, our method can scale to training on 10k hours of robot data and match the performance of diffusion VLAs, while reducing training time by up to 5x.\n[...]\nFigure 4: Overview of the FAST action tokenization pipeline. Given a normalized c",
"source_url": "https://arxiv.org/abs/2501.09747",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2501.09747",
"_exa_published_date": "2025-01-16T00:00:00.000Z"
},
{
"title": "FAST: Efficient Action Tokenization for Vision-Language ... - arXiv",
"snippet": "Autoregressive sequence models, such as Transformer-based vision-language action (VLA) policies, can be tremendously effective for capturing complex and generalizable robotic behaviors. However, such models require us to choose a tokenization of our continuous action signals, which determines how the discrete symbols predicted by the model map to continuous robot actions. We find that current approaches for robot action tokenization, based on simple per-dimension, per-timestep binning schemes, typically perform poorly when learning dexterous skills from high-frequency robot data. To address this challenge, we propose a new compression-based tokenization scheme for robot actions, based on the discrete cosine transform. Our tokenization approach, Frequency-space Action Sequence Tokenization (FAST), enables us to train autoregressive VLAs for highly dexterous and high-frequency tasks where standard discretization methods fail completely. Based on FAST, we release FAST+, a universal robot action tokenizer, trained on 1M real robot action trajectories. It can be used as a black-box tokenizer for a wide range of robot action sequences, with diverse action spaces and control frequencies. Finally, we show that, when combined with the $\\bm{\\pi_{0}}$ VLA, our method can scale to training on 10k hours of robot data and match the performance of diffusion VLAs, while reducing training time by up to 5x.\n[...]\nFigure 4: Overview of the FAST action tokenization pipeline. Given a normalized c",
"source_url": "https://arxiv.org/html/2501.09747v1",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2501.09747v1",
"_exa_published_date": "2025-01-16T00:00:00.000Z"
},
{
"title": "ActionCodec: What Makes for Good Action Tokenizers",
"snippet": "without any robotics\n[...]\nintroduce ActionCodec, a robust action tokenizer that integrates the\n[...]\n. Moreover, ActionCodec leverages Residual Vector Quantization (RVQ) (Lee et al., 2022) post-training to refine reconstruction fidelity and incorporates embodiment-specific soft prompts to facilitate knowledge transfer across diverse robotic platforms\n[...]\nthat VLA\n[...]\nActionCodec, without any additional architectural modifications\n[...]\nefficiency, success rates\n[...]\n. ActionCodec achieves SOTA performance in both\n[...]\nenvironments, providing a systematic\n[...]\nfor the future of VQ-\n[...]\nOur contributions are\n[...]\nas follows:\n[...]\nTokenization Schemes\n[...]\n20\n[...]\nsuffers from low training efficiency and ignores the\n[...]\nparallel decoding (\n[...]\n(Goy\n[...]\n., 2025),\n[...]\nfundamental inefficiencies of heuristic binning. Other\n[...]\nrepresent actions as strings for direct VLM prediction (Hancock et\n[...]\nhowever, this approach\n[...]\nno significant performance benefits while greatly increasing the token budget and extending latency to several seconds, limiting\n[...]\nPertsch et al., 2025) introduces Byte-Pair Encoding (BPE) on frequency-domain signals; however, its reliance on fixed geometric priors limits its capacity for cross-embodiment knowledge transfer. Data-driven schemes, particularly those based on Vector Quantization (VQ) (Wang et al., 2025b; Belkhale and Sadigh, 2024; Mete et al., 2024; Lee et al., 2024), offer a more flexible alternative by learning disc",
"source_url": "https://arxiv.org/abs/2602.15397",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2602.15397",
"_exa_published_date": null
},
{
"title": "OAT: Ordered Action Tokenization",
"snippet": "action tokenization\n[...]\nTo bridge this gap, we introduce Ordered Action Tokenization (OAT), a learned action tokenizer that discretizes continuous action chunks into highly compressed and causally ordered token sequences. OAT employs transformer-based register tokens to aggregate temporal information, finite scalar quantization (FSQ) to construct a discrete bottleneck, and nested dropout to explicitly induce ordering that aligns the latent space with autoregressive generation. The resulting tokenization ensures that any token prefix corresponds to a plausible action chunk. Beyond improved modelability, the ordered structure learned by OAT enables a key capability absent from prior approaches: prefix-based decoding. Autoregressive policies may terminate generation early and still produce valid actions, yielding a natural trade-off between computation and action fidelity. As additional tokens are generated, decoded actions are progressively refined.\n[...]\nAn alternative line of work explores frequency-domain compression, for instance Frequency-space Action Sequence Tokenization (FAST) [49], which employs the Discrete Cosine Transform (DCT) to decompose action chunks into frequency coefficients, followed by Byte Pair Encoding (BPE) [18]. FAST achieves high information density (P.1), and crucially, its low-frequency components first then high-frequency components ordering (P.3) improves downstream autoregressive policies: early token predictions capture the overall trajectory s",
"source_url": "https://arxiv.org/html/2602.04215",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2602.04215",
"_exa_published_date": null
},
{
"title": "PD-VLA: Accelerating Vision-Language-Action Model Integrated with Action Chunking via Parallel Decoding",
"snippet": "the above challenges, we present a novel parallel decoding framework for the mainstream VLA model with action chunking, called Parallel\n[...]\nfor VLA (PD-VLA). Fig. 1 illustrates the\n[...]\nconcept of our parallel decoding approach. Our\n[...]\naction decoding as a system of\n[...]\nsolved through parallel fixed-point iteration methods, e.g., Jacobi fix-point iteration method [32]. This approach preserves\n[...]\nimproving decoding speed\n[...]\nthat we only accelerate the decoding process\n[...]\nVLA inference.\n[...]\n, our method enables friendly\n[...]\ntraining-free acceleration without redesign and modification of models (\n[...]\n, our method\n[...]\nsynergy with existing acceleration\n[...]\nVarious acceleration strategies, including quantization [21] and token pruning [5], have been effectively applied to LLMs, yet they often fail to meet the stringent real-time requirements of action generation. Efforts to enhance efficiency have led to architectural modifications in VLA models, such as DeeR-VLA [43], which dynamically adjusts inference depth, and QAIL [33], which integrates quantization-aware training. Further innovations, like RoboMamba [25] and TinyVLA [41], replace traditional attention mechanisms or focus on developing lightweight models from the ground up, frequently necessitating model re-training and additional data collection. Meanwhile, VLA-Cache [42] selectively caches static tokens and recomputes only dynamic or task-relevant ones. FAST [34] proposes a compression-based toke",
"source_url": "https://arxiv.org/html/2503.02310v2",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2503.02310v2",
"_exa_published_date": null
},
{
"title": "Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware | OpenReview",
"snippet": "Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware | OpenReview\n[...]\n## Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware\n[...]\nAbstract: Fine manipulation tasks, such as threading cable ties or slotting a battery, are notoriously difficult for robots because they require precision, careful coordination of contact forces, and closed-loop visual feedback. Performing these tasks typically requires high-end robots, accurate sensors, or careful calibration, which can be expensive and difficult to set up. Can learning enable low-cost and imprecise hardware to perform these fine manipulation tasks? We present a low-cost system that performs end-to-end imitation learning directly from real demonstrations, collected with a custom teleoperation interface. Imitation learning, however, presents its own challenges, particularly in high-precision domains: errors in the policy can compound over time, and human demonstrations can be non-stationary. To address these challenges, we develop a simple yet novel algorithm, Action Chunking with Transformers (ACT), which learns a generative model over action sequences. ACT allows the robot to learn 6 difficult tasks in the real world, such as opening a translucent condiment cup and slotting a battery with 80-90% success, with only 10 minutes worth of demonstrations.",
"source_url": "https://openreview.net/forum?id=e8Eu1lqLaf",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/forum?id=e8Eu1lqLaf",
"_exa_published_date": "2023-07-09T06:52:40.000Z"
},
{
"title": "Learning Fine-Grained Bimanual Manipulation with Low-Cost ... - arXiv",
"snippet": "Fine manipulation tasks, such as threading cable ties or slotting a battery, are notoriously difficult for robots because they require precision, careful coordination of contact forces, and closed-loop visual feedback. Performing these tasks typically requires high-end robots, accurate sensors, or careful calibration, which can be expensive and difficult to set up. Can learning enable low-cost and imprecise hardware to perform these fine manipulation tasks? We present a low-cost system that performs end-to-end imitation learning directly from real demonstrations, collected with a custom teleoperation interface. Imitation learning, however, presents its own challenges, particularly in high-precision domains: errors in the policy can compound over time, and human demonstrations can be non-stationary. To address these challenges, we develop a simple yet novel algorithm, Action Chunking with Transformers (ACT), which learns a generative model over action sequences. ACT allows the robot to learn 6 difficult tasks in the real world, such as opening a translucent condiment cup and slotting a battery with 80-90% success, with only 10 minutes worth of demonstrations. Project website: tonyzhaozh.github.io/aloha\n[...]\nImitation learning algorithm. Tasks that require precision and visual feedback present a significant challenge for imitation learning, even with high-quality demonstrations. Small errors in the predicted action can incur large differences in the state, exacerbating the “co",
"source_url": "https://arxiv.org/abs/2304.13705",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/abs/2304.13705",
"_exa_published_date": "2023-04-23T00:00:00.000Z"
},
{
"title": "Learning Bimanual Manipulation via Action Chunking and Inter-Arm Coordination with Transformers",
"snippet": "coordinated biman\n[...]\n. To address the\n[...]\narms, particularly for synchronized actions. Therefore, we propose a novel imitation learning architecture that predicts cooperative actions. We differentiate the architecture for both arms and add an intermediate encoder layer, Inter-Arm Coordinated transformer Encoder (IACE),\n[...]\nInter-Arm Coordinated transformer Encoder (IACE), that can adjust the synchronization and timing of potential bimanual movements against the encoders corresponding to each arm. Our overall model\n[...]\na local Transformer encoder for each\n[...]\narm trajectory, the IACE to facilitate learning biman\n[...]\nactions, and a Transformer decoder to\n[...]\nthe action chunk. We compare two types of Transformer decoders: split decoders and single decoders.\n[...]\nWe build our proposed models on the ACT model to design different encoder and decoder structures. In particular, we propose a new design called the inter-arm coordinated transformer Encoder (IACE), which helps synchronize and time the movements of both arms.\n[...]\npropose basic architectures that consist of encoders\n[...]\narm, designed to leverage the biman\n[...]\nfeatures the IACE, allowing the individual robot arms to learn their trajectories while simultaneously considering the state\n[...]\nThe model should focus on the corresponding wrist camera and joint values to determine the appropriate trajectory for each robot arm. Each arm is supported by its local encoder. Global information is also integrated t",
"source_url": "https://arxiv.org/html/2503.13916",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2503.13916",
"_exa_published_date": null
},
{
"title": "ALPHA-𝛼 and Bi-ACT Are All You Need: Importance of Position and Force Information/Control for Imitation Learning of Unimanual and Bimanual Robotic Manipulation with Low-Cost System",
"snippet": "Autonomous manipulation in everyday tasks requires flexible action generation to handle complex, diverse real-world environments, such as objects with varying hardness and softness. Imitation Learning (IL) enables robots to learn complex tasks from expert demonstrations. However, a lot of existing methods rely on position/unilateral control, leaving challenges in tasks that require force information/control, like carefully grasping fragile or varying-hardness objects. As the need for diverse controls increases, there are demand for low-cost bimanual robots that consider various motor inputs. To address these challenges, we introduce Bilateral Control-Based Imitation Learning via Action Chunking with Transformers(Bi-ACT) and”A” ”L”ow-cost ”P”hysical ”Ha”rdware Considering Diverse Motor Control Modes for Research in Everyday Bimanual Robotic Manipulation (ALPHA- $\\alpha$ ). Bi-ACT leverages bilateral control to utilize both position and force information, enhancing the robots adaptability to object characteristics such as hardness, shape, and weight. The concept of ALPHA- $\\alpha$ is affordability, ease of use, repairability, ease of assembly, and diverse control modes (position, velocity, torque), allowing researchers/developers to freely build control systems using ALPHA- $\\alpha$ . In our experiments, we conducted a detailed analysis of Bi-ACT in unimanual manipulation tasks, confirming its superior performance and adaptability compared to Bi-ACT without force control. Base",
"source_url": "https://arxiv.org/html/2411.09942",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://arxiv.org/html/2411.09942",
"_exa_published_date": null
},
{
"title": "[PDF] Learning Fine-Grained Bimanual Manipulation with Low-Cost ...",
"snippet": "W7/A/\u0003cs^u\"Nf|5\n[...]\nǨ,~.a(Ц\u0007\u001ePb%\u0019\u001a?3\u001eS~l\u0012j\u0012tf;m3\n[...]\n7h\n[...]\n4\u001e\u0013L\n[...]\n\u000f9\\ \u0001\u001a 'TG15f-ZB34\u0000[\u0006]ӦbQiuK\u0003g4 h⠯ 8@SB\u0002˔.X\u0007IJ)H\u0007IA\u0011Wd\u001a\n[...]\nZ\u0002O[\u0018^jM\u000f\u001dwEdf\u0014hC\u0004\\xͫ\u0010SB\b7\u001aNbܫ#~87MD\n[...]\n\u0001\n[...]\nʬ`y11\u001fB׶fR\u001fT$,Υ;lv\u0015@+%\u0000O;kS6\"\u0002#J\n[...]\nB2+\u0017.\u0004i\u0015C \u0005 \u0003aGT\u001bW\u0015\u000eגRM;\u0004?9$\u0014&\u0013zTɅz=7-ip1S3 +\u001fI\u0013q^\u0018:.3,\"tv\u001c(c2Mf~\u0010'\u001cӝӨBfd&\u0004\u001d\u00042\b C\u001e\u0006SΩTA8M!Ulŏf-\u0003\u0004ݩF30nR]E\b\u0003]\\r\\n5 QuϬZZ|\u0003SD<_(DU<80x)\u0006phT\u0016L\u0006וX#n^+^q8e0\n[...]\nU*cUOB\u0011:\u0013 *. D:\u001biځ\u0015 l:&:Ybxހl̰?\u00149M ^|ʛ1H6|p37I\u0002\u0001Z棍\u0010B!\\A\u0012KdrT5>ƨ^t\u001e\\|m\u0011>poŅ\u0012#5%w[|+\u0016iXfyXޕݔsM\u0014:91b7\u0015=U\n[...]\n\u0012m\u001d! I4\n[...]\nA<ᯟ \u0006\u0018&x⥌;H ,#n\u0013U?xؔd\u0001]8HoJ}\u0007 ܓ =Z\u0005GT?\u0004㘨\u0007\u001a!5[xDi\u001cGc\\\u000f%2c\u001bd6=dSm;\u0004ݏF\u001e\b1T\u0006 \u0012ch\u0007 \u000f\bV^ԥe_P, \u0010h_O\u0012ٱj\u0004J\u001f?\u0012\u0012y\u0017 :{Zs\u001bx\u001d \\_W!\u0006ЬpU!\u001dlmnBE\u0018g\u0012\u0014wɺ8\u0010N\u0011vM`\u001dYD|=ZI8E\u001c~\\Ϫ1K(\u0006\u0005&\u0015\u0010R\u0012_\u001e\u0000F{ \u001bm&u\\\u0005ݶNg\u00114!\u0005Kh.a3o.'2\\󷉏wi3 >)iJsR 3 FBO\u00187\u0019劍o s,*P\u0012\u001e\u0018\"`G\u000f2l$qYO?r\u001e_P{\u0001ܷG\b `6\u001fz.+mt%9#E!$JtJe$8C6ͥ\u001eÙOV]bnvi)u0fHy'\u0010a_,DM&uǒ|s]D%E>\"\"6 $ ACQ\u0003}\u0011 K/bD:ЗG<\u001cy+wj|U\u0000r\u0016Le5GN|\u0010K Yv`\u0000K J0",
"source_url": "https://openreview.net/pdf/4abc35d9793e56c5b73634eaf903e2495311fbcf.pdf",
"discovered_for": [
"rw.vla"
],
"_exa_id": "https://openreview.net/pdf/4abc35d9793e56c5b73634eaf903e2495311fbcf.pdf",
"_exa_published_date": null
},
{
"title": "[1512.03385] Deep Residual Learning for Image Recognition - arXiv",
"snippet": "[1512.03385] Deep Residual Learning for Image Recognition\n[...]\n# Deep Residual Learning for Image Recognition\n[...]\nKaiming He Xiangyu Zhang Shaoqing Ren Jian Sun Microsoft Research {kahe, v-xiangz, v-shren, jiansun}@microsoft.com\n[...]\nDeeper neural networks are more difficult to train. We present a residual learning framework to ease the training of networks that are substantially deeper than those used previously. We explicitly reformulate the layers as learning residual functions with reference to the layer inputs, instead of learning unreferenced functions. We provide comprehensive empirical evidence showing that these residual networks are easier to optimize, and can gain accuracy from considerably increased depth. On the ImageNet dataset we evaluate residual nets with a depth of up to 152 layers—8 $\\times$ deeper than VGG nets [41] but still having lower complexity. An ensemble of these residual nets achieves 3.57% error on the ImageNet test set. This result won the 1st place on the ILSVRC 2015 classification task. We also present analysis on CIFAR-10 with 100 and 1000 layers.\n[...]\nreferenced mapping. To\n[...]\nextreme, if\n[...]\nwould be easier\n[...]\nresidual to zero than\n[...]\nnonlinear layers.\n[...]\non ImageNet [36\n[...]\nOn the ImageNet classification dataset [36], we obtain excellent results by extremely deep residual nets. Our 152-layer residual net is the deepest network ever presented on ImageNet, while still having lower complexity than VGG nets [41]. Our ensem",
"source_url": "https://arxiv.org/abs/1512.03385",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/1512.03385",
"_exa_published_date": "2015-12-10T00:00:00.000Z"
},
{
"title": "[1512.03385] Deep Residual Learning for Image Recognition",
"snippet": "[1512.03385] Deep Residual Learning for Image Recognition\n[...]\n# Deep Residual Learning for Image Recognition\n[...]\nKaiming He Xiangyu Zhang Shaoqing Ren Jian Sun Microsoft Research {kahe, v-xiangz, v-shren, jiansun}@microsoft.com\n[...]\nDeeper neural networks are more difficult to train. We present a residual learning framework to ease the training of networks that are substantially deeper than those used previously. We explicitly reformulate the layers as learning residual functions with reference to the layer inputs, instead of learning unreferenced functions. We provide comprehensive empirical evidence showing that these residual networks are easier to optimize, and can gain accuracy from considerably increased depth. On the ImageNet dataset we evaluate residual nets with a depth of up to 152 layers—8 $\\times$ deeper than VGG nets [41] but still having lower complexity. An ensemble of these residual nets achieves 3.57% error on the ImageNet test set. This result won the 1st place on the ILSVRC 2015 classification task. We also present analysis on CIFAR-10 with 100 and 1000 layers.\n[...]\nreferenced mapping. To\n[...]\nextreme, if\n[...]\nwould be easier\n[...]\nresidual to zero than\n[...]\nnonlinear layers.\n[...]\non ImageNet [36\n[...]\nOn the ImageNet classification dataset [36], we obtain excellent results by extremely deep residual nets. Our 152-layer residual net is the deepest network ever presented on ImageNet, while still having lower complexity than VGG nets [41]. Our ensem",
"source_url": "https://arxiv.org/abs/1512.03385v1",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/1512.03385v1",
"_exa_published_date": null
},
{
"title": "[PDF] Deep Residual Learning for Image Recognition - People | MIT CSAIL",
"snippet": "Deep Residual Learning\nfor Image Recognition\nKaiming He, Xiangyu Zhang, Shaoqing Ren, Jian Sun\nwork done at\nMicrosoft Research Asia\n[...]\nResNet @ ILSVRC & COCO 2015 Competitions\n[...]\n1st places in all five main tracks\n[...]\n• ImageNet Classification: “Ultra-deep” 152-layer nets \n• ImageNet Detection: 16% better than 2nd\n[...]\n• ImageNet Localization: 27% better than 2nd\n[...]\n• COCO Detection: 11% better than 2nd\n[...]\n• COCO Segmentation: 12% better than 2nd\n[...]\n*improvements are relative numbers\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nRevolution of Depth\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiangyu Zhang, Shaoqing Ren, & Jian Sun. “Deep Residual Learning for Image Recognition”. CVPR 2016.\n[...]\nKaiming He, Xiang",
"source_url": "https://pdfs.semanticscholar.org/1cea/9b1931b9e87641708fec43d03f2a58f4d2b0.pdf",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://pdfs.semanticscholar.org/1cea/9b1931b9e87641708fec43d03f2a58f4d2b0.pdf",
"_exa_published_date": null
},
{
"title": "Deep Residual Learning for Image Recognition: A Survey - MDPI",
"snippet": "Deep Residual Learning for Image Recognition: A Survey\n[...]\n# Deep Residual Learning for Image Recognition: A Survey\n[...]\nMuhammad Shafiq\n[...]\n1,* and\n\nZhaoquan Gu\n[...]\n2,3,*\n[...]\nCyberspace Institute of Advanced Technology, Guangzhou University, Guangzhou 510006, China\n[...]\nDepartment of New Networks, Peng Cheng Laboratory, Shenzhen 518055, China\n[...]\nDepartment of Computer Science and Technology, Harbin Institute of Technology, Shenzhen 518055, China\n[...]\nAppl. Sci. 2022, 12(18), 8972; https://doi.org/10.3390/app12188972\n[...]\nDeep Residual Networks have recently been shown to significantly improve the performance of neural networks trained on ImageNet, with results beating all previous methods on this dataset by large margins in the image classification task. However, the meaning of these impressive numbers and their implications for future research are not fully understood yet. In this survey, we will try to explain what Deep Residual Networks are, how they achieve their excellent results, and why their successful implementation in practice represents a significant advance over existing techniques. We also discuss some open questions related to residual learning as well as possible applications of Deep Residual Networks beyond ImageNet. Finally, we discuss some issues that still need to be resolved before deep residual learning can be applied on more complex problems.\n[...]\ndeep residual learning for image recognition\n[...]\ndeep residual learning;\n[...]\nDeep resid",
"source_url": "https://www.mdpi.com/2076-3417/12/18/8972",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://www.mdpi.com/2076-3417/12/18/8972",
"_exa_published_date": null
},
{
"title": "[1603.05027] Identity Mappings in Deep Residual Networks - arXiv",
"snippet": "Kaiming He Xiangyu Zhang Shaoqing Ren Jian Sun\n[...]\nDeep residual networks [1] have emerged as a family of extremely deep architectures showing compelling accuracy and nice convergence behaviors. In this paper, we analyze the propagation formulations behind the residual building blocks, which suggest that the forward and backward signals can be directly propagated from one block to any other block, when using identity mappings as the skip connections and after-addition activation. A series of ablation experiments support the importance of these identity mappings. This motivates us to propose a new residual unit, which makes training easier and improves generalization. We report improved results using a 1001-layer ResNet on CIFAR-10 (4.62% error) and CIFAR-100, and a 200-layer ResNet on Image\n[...]\n. Code is available at: https://github.com/KaimingHe/resnet-1k-layers.\n[...]\n- [1] He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR. (2016)",
"source_url": "https://arxiv.org/abs/1603.05027",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/1603.05027",
"_exa_published_date": null
},
{
"title": "",
"snippet": "(AttnRes) [30] recently showed that the fixed residual connections in Trans\u0002formers, which accumulate layer outputs with uniform unit weights, can be re\u0002placed by learned softmax attention over all preceding layer outputs. A single\n[...]\npseudo-query vector per layer selects which earlier representations to aggregate,\n[...]\nenabling content-aware, position-specific routing with minimal overhead. This\n[...]\nprinciple of learned\n[...]\nAttention Residuals (AttnRes) were proposed by Chen et al. [30] for Transformer\u0002based LLMs. The standard residual connection accumulates layer outputs with\n[...]\nfixed unit weights: hl = hl1 + fl(hl1). As network depth grows, this causes\n[...]\ntwo problems: feature dilution, where each layers relative contribution to the\n[...]\naccumulated sum diminishes, and unbounded magnitude growth, a well-known\n[...]\nissue in PreNorm Transformers. AttnRes replaces the fixed accumulation with\n[...]\nsoftmax attention over all preceding layer outputs, parameterized by a single\n[...]\nlearned pseudo-query wl ∈ R\n[...]\nd per layer. This enables selective, content-aware\n[...]\naccess to earlier representations while keeping output magnitudes bounded.\n[...]\nWe extend Attention Residuals from same-dimensional layer aggregation in\n[...]\nIn standard Transformers [31], the residual connection at layer l accumulates\n[...]\nwhere fl denotes the layer computation (e.g., self-attention or feed-forward net\u0002work). Attention Residuals [30] replace this fixed accumulation with a",
"source_url": "https://arxiv.org/pdf/2604.03297",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/pdf/2604.03297",
"_exa_published_date": null
},
{
"title": "[2603.15031] Attention Residuals - arXiv",
"snippet": "Untitled Document\n\n$0$ $5$ $10$ $15$ $20$\n\nUntitled Document\n$0$ $5$ $10$ $15$ $20$\nBETA",
"source_url": "https://arxiv.org/abs/2603.15031",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/2603.15031",
"_exa_published_date": "2026-03-16T00:00:00.000Z"
},
{
"title": "SiameseNorm: Breaking the Barrier to Reconciling Pre/Post-Norm",
"snippet": "In this paper, we propose SiameseNorm, an elegant two\n[...]\nstream residual architecture that unifies\n[...]\nmaintain two residual streams\n[...]\nshared parameters:\n[...]\nthe advantages of\n[...]\nnegligible computational overhead\n[...]\nboosts accuracy from\n[...]\n128\n[...]\n639.6\n[...]\nwhere the product denotes an ordered composition of Jacobians from layer N1N-1 down to i+1i+1. Notably, the term 𝐈\\mathbf{I} corresponds to the skip connection, which preserves an explicit identity gradient path. This allows gradients to flow through the network without explicit attenuation, facilitating the training of large scale models. However, it implicitly allows the representation magnitudes to grow unbounded. As noted previously, Pre-Norm exhibits insufficient effective depth, an issue that likely stems from a structural mismatch: As shown in Figure˜2(a), the main path accumulates residual updates without re-normalization, causing hidden state magnitudes to grow with depth (peri-ln). Consequently, deeper blocks encounter a scaling imbalance: they must influence an increasingly high-magnitude main path while being restricted to normalized, fixed-scale inputs. This growing disparity effectively dilutes the relative contribution of deeper layers, thereby limiting the effective depth of the model.\n[...]\nBy maintaining a clean identity path, Pre-Norm ensures stable gradient propagation. However, this comes at the cost of unbounded magnitude growth. As illustrated in Figure 2(a), while the input ",
"source_url": "https://arxiv.org/html/2602.08064v1",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/html/2602.08064v1",
"_exa_published_date": null
},
{
"title": "[PDF] Attention Residuals - arXiv",
"snippet": "Untitled Document\n\n$0$ $5$ $10$ $15$ $20$\n\nUntitled Document\n$0$ $5$ $10$ $15$ $20$\nBETA",
"source_url": "https://arxiv.org/pdf/2603.15031",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/pdf/2603.15031",
"_exa_published_date": "2026-03-16T00:00:00.000Z"
},
{
"title": "",
"snippet": "However, deep PreNorm models exhibit their own limita\u0002tions. The clean residual path can lead to representation\n[...]\nlayers fail to learn new features (Li\n[...]\n). This “Curse of Depth” is characterized by\n[...]\nexponential growth in activation variance and layers that de\u0002volve into identity functions, limiting the benefits of scaling\n[...]\nPre-LayerNorm (PreNorm): To address the training in\u0002stability of PostNorm, especially when training deep models,\n[...]\n. Here,\n[...]\neach sub-layer,\n[...]\n. This creates a clean, identity\n[...]\nover PreNorm\n[...]\nRepresentation Collapse A primary pathology in deep\n[...]\nPreNorm Transformers is the uncontrolled growth of fea\u0002ture variance, which leads to representation collapse. In a\n[...]\nstandard PreNorm block, the main residual path acts as an\n[...]\nidentity map, causing variance to accumulate linearly with\n[...]\ndepth. Formally, for a network of depth L, the variance of\n[...]\nthe hidden states scales as Var(X\n[...]\nΘ(L) (Kedia et al.,\n[...]\n024).\n[...]\nThis variance explosion degrades the learning capability of\n[...]\ndeep layers. Consider the Jacobian of the l-th PreNorm\n[...]\nblock. Let Res(·) denote the transformation within the resid\u0002ual branch (e.g., Self-Attention or FFN). Since the transfor\u0002mation within the residual branch operates on inputs nor\u0002malized by the feature standard deviation σl =\n[...]\nthe Jacobian JPre can be expressed as:\n[...]\nAs l → ∞, σl → ∞, causing the residual term to vanish at a\n[...]\nrate of O(1/\n[...]\nl).",
"source_url": "https://arxiv.org/pdf/2601.22580v1",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/pdf/2601.22580v1",
"_exa_published_date": null
},
{
"title": "[2409.19606] Hyper-Connections - arXiv",
"snippet": "We present hyper-connections, a simple yet effective method that can serve as an alternative to residual connections. This approach specifically addresses common drawbacks observed in residual connection variants, such as the seesaw effect between gradient vanishing and representation collapse. Theoretically, hyper-connections allow the network to adjust the strength of connections between features at different depths and dynamically rearrange layers. We conduct experiments focusing on the pre-training of large language models, including dense and sparse models, where hyper-connections show significant performance improvements over residual connections. Additional experiments conducted on vision tasks also demonstrate similar improvements. We anticipate that this method will be broadly applicable and beneficial across a wide range of AI problems.\n[...]\nDeep learning has achieved tremendous success across various domains, where residual connections (He et al., 2016) have been instrumental in contemporary neural network architectures, including transformers and CNNs. Residual connections help mitigate the problem of gradient vanishing, enabling the effective training of very deep networks. However, it is important to acknowledge that residual connections are not infallible solutions and still present limitations that remain unresolved.\n[...]\nDriven by the limitations of residual connections, an important question arises: Can neural networks autonomously learn the optimal streng",
"source_url": "https://arxiv.org/abs/2409.19606",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://arxiv.org/abs/2409.19606",
"_exa_published_date": "2024-09-29T00:00:00.000Z"
},
{
"title": "Hyper-Connections - OpenReview",
"snippet": "Hyper-Connections | OpenReview\n\n## Hyper-Connections\n\n### Defa Zhu, Hongzhi Huang, Zihao Huang, Yutao Zeng, Yunyao Mao, Banggu Wu, Qiyang Min, Xun Zhou\n\nICLR 2025 Postereveryonesince 04 Oct 2024\">Everyone Revisions BibTeX CC BY 4.0\n\nKeywords: Network Architecture, Residual Connections, LLMs, Pre-training\n\nAbstract: We present hyper-connections, a simple yet effective method that can serve as an alternative to residual connections. This approach specifically addresses common drawbacks observed in residual connection variants, such as the seesaw effect between gradient vanishing and representation collapse. Theoretically, hyper-connections allow the network to adjust the strength of connections between features at different depths and dynamically rearrange layers. We conduct experiments focusing on the pre-training of large language models, including dense and sparse models, where hyper-connections show significant performance improvements over residual connections. Additional experiments conducted on vision tasks also demonstrate similar improvements. We anticipate that this method will be broadly applicable and beneficial across a wide range of AI problems.\n\nPrimary Area: foundation or frontier models, including LLMs\n\nCode Of Ethics: I acknowledge that I and all co-authors of this work have read and commit to adhering to the ICLR Code of Ethics.\n\nSubmission Guidelines: I certify that this submission complies with the submission instructions as described on https://iclr.cc/Con",
"source_url": "https://openreview.net/forum?id=9FqARW7dwB",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://openreview.net/forum?id=9FqARW7dwB",
"_exa_published_date": null
},
{
"title": "Birkhoff-Exact Hyper-Connections: Exact Spectral Stability for Deep Residual Networks | OpenReview",
"snippet": "## Birkhoff-Exact Hyper-Connections: Exact Spectral Stability for Deep Residual Networks\n[...]\nKeywords: doubly stochastic matrices, spectral stability, deep residual networks, Birkhoff-von Neumann theorem, hyper-connections, token mixing, extreme depth training, quantization robustness\n[...]\nTL;DR: We propose BE-HC, which uses the Birkhoff-von Neumann theorem to construct exactly doubly stochastic mixing matrices as convex combinations of permutation matrices, enabling stable training at 1000+ layers where prior methods fail.\n[...]\nAbstract: Learnable information routing in deep networks faces the *depth-stability-efficiency trilemma*: architectures that scale to extreme depths often sacrifice efficiency; efficient approaches lack stability guarantees. Prior work uses iterative Sinkhorn-Knopp normalization to approximate doubly stochastic mixing matrices, but residual errors destabilize training beyond several hundred layers. We propose **Birkhoff-Exact Hyper-Connections (BE-HC)**, which leverages the Birkhoff-von Neumann theorem to construct *exactly* doubly stochastic matrices as convex combinations of permutation matrices. This guarantees spectral radius $\\rho = 1$ exactly—not approximately—enabling stable training at unprecedented depths. **Key results:** (1) *Extreme depth:* BE-HC trains stably at **1000 layers**, achieving 35.71% accuracy where ReZero and other baselines fail to converge. (2) *Long context:* BE-HC handles **8K tokens** on a single V100 GPU (22.56% vali",
"source_url": "https://openreview.net/forum?id=jpIjkN1B1Q",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://openreview.net/forum?id=jpIjkN1B1Q",
"_exa_published_date": "2026-03-02T22:01:32.000Z"
},
{
"title": "",
"snippet": "We present the first study of Hyper-Connections (HC) for volumetric multi-modal\n[...]\nbrain tumor segmentation, integrating them as a drop-in replacement for fixed\n[...]\nresidual connections across five architectures: nnU-Net, SwinUNETR, VT-UNet,\n[...]\nNetpp.\n[...]\nIn this work, we explore Hyper-Connections (HC) [18], a recently proposed generalization of residual\n[...]\nconnections that enables dynamic, input-dependent feature aggregation. While HC has shown\n[...]\nit has not been investigated for medical image segmentation, particularly in volumetric and multi\u0002modal settings. We extend HC\n[...]\nboth 3D\n[...]\nand training dynamics. Residual connections [5]\n[...]\nstable optimization of deep networks through\n[...]\narchitectural components. In contrast, the Hyper-Connection framework [18] provides a unified\n[...]\nformulation that simultaneously adapts both depth-wise and width-wise connections, representing a\n[...]\nstrict generalization of prior approaches. To the best of our knowledge, this work presents the first\n[...]\nHyper-Connections (HC) were originally proposed in the context of large language model pre-training,\n[...]\nwhere the primary challenge is to maintain stable gradient flow across very deep transformer archi\u0002tectures operating on sequential token embeddings [18]. In that setting, HC improves optimization\n[...]\nby learning adaptive depth-wise aggregation, thereby avoiding limitations associated with fixed\n[...]\nresidual formulations such as Pre-Norm and Post-Norm va",
"source_url": "https://www.arxiv.org/pdf/2603.19844",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://www.arxiv.org/pdf/2603.19844",
"_exa_published_date": null
},
{
"title": "Ablate and Rescue: A Causal Analysis of Residual Stream Hyper-Connections",
"snippet": "Multi-stream transformer architectures have recently been proposed as a promising direction for managing representation collapse and the vanishing gradient problem for residual connections, yet their internal mechanisms remain unexplored. In particular, the recently introduced Manifold-Constrained Hyper-Connections (mHC) architecture posits multiple residual streams with constrained interaction, but lacks in-depth mechanistic analysis. We present the first open-source mHC language model (https://huggingface.co/wgpeng/mhc-780m) and analyze the multiple-stream architecture with a suite of representation-level metrics and causal interventions to probe how parallel streams encode and utilize information. Specifically, we introduce a systematic stream ablation-and-rescue framework that enables direct causal comparison of residual streams during inference. Through targeted pairwise interventions and controlled recovery experiments, we distinguish functional redundancy from asymmetric utilization and reveal how information is distributed across streams beyond what is observable from representational similarity alone.\n[...]\nHyper-Connections extend the standard transformer residual architecture by allowing multiple residual streams per layer, dynamically mixed through learned routing matrices (He et al., 2015; Zhu et al., 2025). Manifold-Constrained Hyper-Connections (mHC) further refines this framework by imposing geometric constraints on inter-stream mixing (Xie et al., 2026).\n[...",
"source_url": "https://www.arxiv.org/pdf/2603.14833",
"discovered_for": [
"rw.attnres"
],
"_exa_id": "https://www.arxiv.org/pdf/2603.14833",
"_exa_published_date": null
},
{
"title": "",
"snippet": "DECOUPLED WEIGHT DECAY REGULARIZATION\n[...]\nIlya Loshchilov & Frank Hutter\n[...]\nL2 regularization and weight decay regularization are equivalent for standard\n[...]\nstochastic gradient descent (when rescaled by the learning rate), but as we demon\u0002strate this is not the case for adaptive gradient algorithms, such as Adam. While\n[...]\nexpose), we propose a simple modification to recover the original formulation of\n[...]\nweight decay regularization by decoupling the weight decay from the optimization\n[...]\nsteps taken w.r.t. the loss function. We provide empirical evidence that our pro\u0002posed modification (i) decouples the optimal choice of weight decay factor from\n[...]\nthe setting of the learning rate for both standard SGD and Adam and (ii) substan\u0002tially improves Adams generalization performance, allowing it to compete with\n[...]\nSGD with momentum on image classification datasets (on which it was previously\n[...]\ntypically outperformed by the latter). Our proposed decoupled weight decay has\n[...]\ncommunity has implemented\n[...]\nit in TensorFlow and PyTorch; the complete source code for our experiments is\n[...]\navailable at https://github.com/loshchil/AdamW-and-SGDW\n[...]\nThe main contribution of this paper is to improve regularization in Adam by decoupling the weight\n[...]\ndecay from the gradient-based update. In a comprehensive analysis, we show that Adam generalizes\n[...]\nsubstantially better with decoupled weight decay than with L2 regularization, achieving 15% relative\n[.",
"source_url": "https://arxiv.org/pdf/1711.05101v3",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/pdf/1711.05101v3",
"_exa_published_date": null
},
{
"title": "[1711.05101] Decoupled Weight Decay Regularization - arXiv",
"snippet": "[1711.05\n[...]\npled Weight Decay Regularization\n[...]\nIlya Loshchilov & Frank Hutter University of Freiburg Freiburg, Germany, {ilya,fh}@cs.uni-freiburg.de\n[...]\nL2 regularization and weight decay regularization are equivalent for standard stochastic gradient descent (when rescaled by the learning rate), but as we demonstrate this is not the case for adaptive gradient algorithms, such as Adam. While common implementations of these algorithms employ L2 regularization (often calling it “weight decay” in what may be misleading due to the inequivalence we expose), we propose a simple modification to recover the original formulation of weight decay regularization by decoupling the weight decay from the optimization steps taken w.r.t. the loss function. We provide empirical evidence that our proposed modification (i) decouples the optimal choice of weight decay factor from the setting of the learning rate for both standard SGD and Adam and (ii) substantially improves Adams generalization performance, allowing it to compete with SGD with momentum on image classification datasets (on which it was previously typically outperformed by the latter). Our proposed decoupled weight decay has already been adopted by many researchers, and the community has implemented it in TensorFlow and PyTorch; the complete source code for our experiments is available at https://github.com/loshchil/AdamW-and-SGDW\n[...]\nThe main contribution of this paper is to improve regularization in Adam by decoupling ",
"source_url": "https://arxiv.org/abs/1711.05101",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/abs/1711.05101",
"_exa_published_date": "2017-11-14T00:00:00.000Z"
},
{
"title": "[PDF] Decoupled Weight Decay Regularization - arXiv",
"snippet": "DECOUPLED WEIGHT DECAY REGULARIZATION\n[...]\nIlya Loshchilov & Frank Hutter\n[...]\nL2 regularization and weight decay regularization are equivalent for standard\n[...]\nstochastic gradient descent (when rescaled by the learning rate), but as we demon\u0002strate this is not the case for adaptive gradient algorithms, such as Adam. While\n[...]\nexpose), we propose a simple modification to recover the original formulation of\n[...]\nweight decay regularization by decoupling the weight decay from the optimization\n[...]\nsteps taken w.r.t. the loss function. We provide empirical evidence that our pro\u0002posed modification (i) decouples the optimal choice of weight decay factor from\n[...]\nthe setting of the learning rate for both standard SGD and Adam and (ii) substan\u0002tially improves Adams generalization performance, allowing it to compete with\n[...]\nSGD with momentum on image classification datasets (on which it was previously\n[...]\ntypically outperformed by the latter). Our proposed decoupled weight decay has\n[...]\ncommunity has implemented\n[...]\nit in TensorFlow and PyTorch; the complete source code for our experiments is\n[...]\navailable at https://github.com/loshchil/AdamW-and-SGDW\n[...]\nThe main contribution of this paper is to improve regularization in Adam by decoupling the weight\n[...]\ndecay from the gradient-based update. In a comprehensive analysis, we show that Adam generalizes\n[...]\nsubstantially better with decoupled weight decay than with L2 regularization, achieving 15% relative\n[.",
"source_url": "https://arxiv.org/pdf/1711.05101",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/pdf/1711.05101",
"_exa_published_date": "2019-01-04T00:00:00.000Z"
},
{
"title": "Decoupled Weight Decay Regularization - OpenReview",
"snippet": "Decoupled Weight Decay Regularization | OpenReview\n\n## Decoupled Weight Decay Regularization\n\nICLR 2019 Conference Blind SubmissionReaders: Everyone\n\nAbstract: L$_2$ regularization and weight decay regularization are equivalent for standard stochastic gradient descent (when rescaled by the learning rate), but as we demonstrate this is \\emph{not} the case for adaptive gradient algorithms, such as Adam. While common implementations of these algorithms employ L$_2$ regularization (often calling it ``weight decay'' in what may be misleading due to the inequivalence we expose), we propose a simple modification to recover the original formulation of weight decay regularization by \\emph{decoupling} the weight decay from the optimization steps taken w.r.t. the loss function. We provide empirical evidence that our proposed modification (i) decouples the optimal choice of weight decay factor from the setting of the learning rate for both standard SGD and Adam and (ii) substantially improves Adam's generalization performance, allowing it to compete with SGD with momentum on image classification datasets (on which it was previously typically outperformed by the latter). Our proposed decoupled weight decay has already been adopted by many researchers, and the community has implemented it in TensorFlow and PyTorch; the complete source code for our experiments is available at \\url{https://github.com/loshchil/AdamW-and-SGDW}\n\nKeywords: optimization, regularization, weight decay, Adam\n\nCode: ",
"source_url": "https://openreview.net/forum?id=Bkg6RiCqY7",
"discovered_for": [
"method.training"
],
"_exa_id": "https://openreview.net/forum?id=Bkg6RiCqY7",
"_exa_published_date": "2018-09-27T08:45:35.000Z"
},
{
"title": "",
"snippet": "Ilya Loshchilov & Frank Hutter \nUniversity of Freiburg \nFreiburg, Germany, \nilya.loshchilov@gmail.com, fh@cs.uni-freiburg.de \n\nABSTRACT\n[...]\nL2 regularization and weight decay regularization are equivalent for standard stochastic gradient descent (when rescaled by the learning rate), but as we demon strate this is not the case for adaptive gradient algorithms, such as Adam. While common implementations of these algorithms employ L2 regularization (often calling it “weight decay” in what may be misleading due to the inequivalence we expose), we propose a simple modification to recover the original formulation of weight decay regularization by decoupling the weight decay from the optimization steps taken w.r.t. the loss function. We provide empirical evidence that our pro posed modification (i) decouples the optimal choice of weight decay factor from the setting of the learning rate for both standard SGD and Adam and (ii) substan tially improves Adams generalization performance, allowing it to compete with SGD with momentum on image classification datasets (on which it was previously typically outperformed by the latter). Our proposed decoupled weight decay has already been adopted by many researchers, and the community has implemented it in TensorFlow and PyTorch; the complete source code for our experiments is available at https://github.com/loshchil/AdamW-and-SGDW\n[...]\n(AdamW),\n[...]\n(Loshchilov & Hutter,\n[...]\n. Figure 1\n[...]\nweight decay (\n[...]\nshow that weight decay ",
"source_url": "https://openreview.net/pdf?id=Bkg6RiCqY7",
"discovered_for": [
"method.training"
],
"_exa_id": "https://openreview.net/pdf?id=Bkg6RiCqY7",
"_exa_published_date": null
},
{
"title": "PyTorch: An Imperative Style, High-Performance Deep Learning Library",
"snippet": "PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\nDeep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it was designed from first principles to support an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several commonly used benchmarks.",
"source_url": "https://papers.nips.cc/paper/2019/hash/bdbca288fee7f92f2bfa9f7012727740-Abstract.html",
"discovered_for": [
"method.training"
],
"_exa_id": "https://papers.nips.cc/paper/2019/hash/bdbca288fee7f92f2bfa9f7012727740-Abstract.html",
"_exa_published_date": null
},
{
"title": "[PDF] An Imperative Style, High-Performance Deep Learning Library - NIPS",
"snippet": "PyTorch: An Imperative Style, High-Performance Deep Learning Library\n\n| | Adam | Paszke | | Sam | Gross | | | Francisco | Massa | | |\n[...]\nAbstract Deep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it provides an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several common benchmarks.\n[...]\nWith the increased interest in deep learning in recent years, there has been an explosion of machine learning tools. Many popular frameworks such as Caffe [1], CNTK [2], TensorFlow [3], and Theano [4], construct a static dataflow graph that represents the computation and which can then be applied repeatedly to batches of data. This approach provides visibility into the whole comput",
"source_url": "https://papers.neurips.cc/paper/9015-pytorch-an-imperative-style-high-performance-deep-learning-library.pdf",
"discovered_for": [
"method.training"
],
"_exa_id": "https://papers.neurips.cc/paper/9015-pytorch-an-imperative-style-high-performance-deep-learning-library.pdf",
"_exa_published_date": null
},
{
"title": "[1912.01703v1] PyTorch: An Imperative Style, High-Performance Deep Learning Library",
"snippet": "[1912.01703v1] PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\n# Title:PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\n> Abstract:Deep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it provides an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several common benchmarks.",
"source_url": "https://arxiv.org/abs/1912.01703v1",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/abs/1912.01703v1",
"_exa_published_date": null
},
{
"title": "PyTorch: An Imperative Style, High-Performance Deep Learning ...",
"snippet": "[1912.01703] PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\n# PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\nAdam Paszke University of Warsaw adam.paszke@gmail.com Sam Gross Facebook AI Research sgross@fb.com Francisco Massa Facebook AI Research fmassa@fb.com Adam Lerer Facebook AI Research alerer@fb.com James Bradbury Google jekbradbury@gmail.com Gregory Chanan Facebook AI Research gchanan@fb.com Trevor Killeen Self Employed killeent@cs.washington.edu Zeming Lin Facebook AI Research zlin@fb.com Natalia Gimelshein NVIDIA ngimelshein@nvidia.com Luca Antiga Orobix luca.antiga@orobix.com Alban Desmaison Oxford University alban@robots.ox.ac.uk Andreas Köpf Xamla andreas.koepf@xamla.com Edward Yang Facebook AI Research ezyang@fb.com Zach DeVito Facebook AI Research zdevito@cs.stanford.edu Martin Raison Nabla martinraison@gmail.com Alykhan Tejani Twitter atejani@twitter.com Sasank Chilamkurthy Qure.ai sasankchilamkurthy@gmail.com Benoit Steiner Facebook AI Research benoitsteiner@fb.com Lu Fang Facebook lufang@fb.com Junjie Bai Facebook jbai@fb.com Soumith Chintala Facebook AI Research soumith@gmail.com\n[...]\nDeep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it provides an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popu",
"source_url": "https://arxiv.org/abs/1912.01703",
"discovered_for": [
"method.training"
],
"_exa_id": "https://arxiv.org/abs/1912.01703",
"_exa_published_date": "2019-12-03T00:00:00.000Z"
},
{
"title": "PyTorch: An Imperative Style, High-Performance Deep Learning Library",
"snippet": "PyTorch: An Imperative Style, High-Performance Deep Learning Library\n[...]\nDeep learning frameworks have often focused on either usability or speed, but not both. PyTorch is a machine learning library that shows that these two goals are in fact compatible: it was designed from first principles to support an imperative and Pythonic programming style that supports code as a model, makes debugging easy and is consistent with other popular scientific computing libraries, while remaining efficient and supporting hardware accelerators such as GPUs. In this paper, we detail the principles that drove the implementation of PyTorch and how they are reflected in its architecture. We emphasize that every aspect of PyTorch is a regular Python program under the full control of its user. We also explain how the careful and pragmatic implementation of the key components of its runtime enables them to work together to achieve compelling performance. We demonstrate the efficiency of individual subsystems, as well as the overall speed of PyTorch on several commonly used benchmarks.",
"source_url": "http://papers.neurips.cc/paper/9015-pytorch-an-",
"discovered_for": [
"method.training"
],
"_exa_id": "http://papers.neurips.cc/paper/9015-pytorch-an-",
"_exa_published_date": null
}
]
}
Binary file not shown.
+269
View File
@@ -0,0 +1,269 @@
\documentclass{article}
\usepackage{corl_2026}
\usepackage[T1]{fontenc}
\usepackage{microtype}
\usepackage{booktabs}
\usepackage{multirow}
\usepackage{graphicx}
\usepackage{amsmath,amssymb}
\usepackage{enumitem}
\usepackage{tcolorbox}
\usepackage[capitalize]{cleveref}
\usepackage{url}
\title{iMF-AttnRes: Fast Mean-Flow Action Generation with Attention Residuals for Simulated Robot Manipulation}
\author{Anonymous Authors}
\begin{document}
\maketitle
\begin{abstract}
Diffusion-style vision-language-action policies provide expressive action distributions, but iterative inference can be too slow for closed-loop manipulation. We study whether improved Mean Flow (iMF), originally motivated by fast-forward generative modeling, can be adapted to action generation so that a policy uses one to a few flow evaluations rather than long denoising chains. We combine iMF with Attention Residuals (AttnRes), which replace fixed residual accumulation in a deep policy transformer with learned depth-wise aggregation over previous layer outputs. In RoboIMI socket peg insertion, the best iMF-AttnRes policy reaches an average reward of 1513.56 and median reward of 1901.5 with 15.122 ms average inference time, compared with diffusion-style infer100 baselines at 861.53/338.291 ms and 1019.39/397.912 ms. In RoboIMI object transfer, the best iMF-AttnRes policy reaches average reward 526.22 and 44/100 success-like episodes, compared with a native diffusion-policy baseline at 319.2 and 29/100. These results suggest that average-flow action generation is a promising route to low-latency robot policies, while also revealing sensitivity to horizon choices, vision-token design, and simulation-to-real validation.
\end{abstract}
\keywords{Robot learning, imitation learning, flow matching, vision-language-action models}
\section{Introduction}
\label{sec:introduction}
Learning-based robot manipulation policies increasingly use expressive generative action models. Diffusion Policy showed that action diffusion can be an effective visuomotor policy class for manipulation~\citep{black2024pi0,chi2023diffusion}, inheriting the representational flexibility of denoising diffusion models~\citep{ho2020denoising} and transformer-based diffusion backbones~\citep{peebles2023scalable}. However, this expressivity often comes with a deployment cost: the policy must be evaluated repeatedly during sampling. In the RoboIMI socket peg experiments studied here, two diffusion-style VLA baselines use infer100 sampling and require 338.291 ms and 397.912 ms per inference on their recorded rollouts. Such latencies are undesirable when the robot must repeatedly close the perception-action loop.
This paper asks a direct question: can a robot action generator retain the quality benefits of generative policies while reducing inference to one or a few evaluations? We explore this question by adapting improved Mean Flow (iMF) to VLA-style imitation learning. Flow matching and rectified flow formulate generation through continuous vector fields~\citep{lipman2023flow,liu2022flow}. Mean Flow reframes generation around average velocity, enabling one-step generation~\citep{geng2025mean}; iMF further addresses instability caused by substituting conditional velocities into nonlinear mean-flow relations~\citep{geng2025improved}. For robot control, this distinction matters because fewer evaluations directly increase control responsiveness.
We pair iMF with Attention Residuals (AttnRes). Residual connections are essential for deep visual and transformer networks~\citep{he2016deep,vaswani2017attention}, but standard residual accumulation adds all layer outputs with fixed unit weight. The motivation in our notes is to rewrite residual networks as a sum over layer increments, then replace the uniform sum with a learned softmax aggregation. AttnRes implements this view by letting each layer attend over preceding residual outputs~\citep{kimi2026attention}. We use this mechanism in the policy transformer to reduce depth-wise feature dilution while keeping the action generator fast.
Our contributions are:
\begin{enumerate}[leftmargin=*]
\item We formulate an iMF-AttnRes VLA policy for simulated robot imitation learning, combining one-to-few-step average-flow action generation with learned residual aggregation.
\item We provide 100-rollout evaluations on two RoboIMI manipulation environments, socket peg insertion and object transfer, comparing against diffusion-style VLA baselines, a native Diffusion Policy baseline, ACT-style action chunking, and SmolVLA-style compact VLA inference~\citep{zhao2023learning,shukor2025smolvla}.
\item We report both task quality and deployment speed. On socket insertion, the strongest iMF-AttnRes run improves average reward over the ph16 diffusion-style baseline by 1.76$\times$ and inference FPS by 22.97$\times$; on object transfer, the strongest iMF-AttnRes run improves average reward over the native diffusion-policy baseline by 1.65$\times$.
\end{enumerate}
We deliberately keep the claims scoped to simulation: hardware varies across runs, and real-robot transfer remains future work.
\section{Related Work}
\label{sec:related_work}
\paragraph{Diffusion and flow policies for robot action generation.}
Diffusion Policy introduced action diffusion as a visuomotor policy learning framework~\citep{black2024pi0,chi2023diffusion}, building on denoising diffusion objectives~\citep{ho2020denoising}. Diffusion transformers further show that transformer backbones can scale generative modeling~\citep{peebles2023scalable}. These models are expressive because they iteratively refine samples, but iterative refinement is also a latency source. Flow matching provides an alternative continuous generative formulation by learning vector fields that transport probability paths~\citep{lipman2023flow}. Rectified flow emphasizes straighter transport paths and faster sampling~\citep{liu2022flow}. Our method follows this fast-generation lineage but uses the improved mean-flow relation to target one-to-few-step action generation.
\paragraph{Vision-language-action models and action chunking.}
Transformer robot policies such as RT-1 and RT-2 show how large sequence models can map observations and task context to robot actions~\citep{brohan2022rt,brohan2023rt}. OpenVLA studies open VLA modeling at scale~\citep{kim2024openvla}, while compact or efficient VLA models such as SmolVLA and FAST reduce deployment cost through smaller backbones or efficient action tokenization~\citep{shukor2025smolvla,pertsch2025fast}. Action Chunking with Transformers (ACT) predicts chunks of future actions to improve temporal consistency and execution efficiency~\citep{zhao2023learning}. The iMF-AttnRes policy is complementary: it retains chunked execution but changes the generative action sampler so that each chunk can be produced with one to a few evaluations.
\paragraph{Mean-flow training for one-step generation.}
Let $x_t$ denote a sample at time $t$ and let the instantaneous flow satisfy
\begin{equation}
\frac{d x_t}{dt} = v(x_t,t).
\end{equation}
For two times $r<t$, Mean Flow defines an average velocity
\begin{equation}
\bar{v}(z_t,r,t) = \frac{1}{t-r}\int_r^t v(z_\tau,\tau)d\tau.
\end{equation}
Differentiating $(t-r)\bar{v}$ with respect to $t$ yields
\begin{equation}
\bar{v}(z_t,r,t)=v(z_t,t) - (t-r)\frac{d}{dt}\bar{v}(z_t,r,t),
\end{equation}
where the total derivative can be written as a Jacobian-vector product (JVP), $\mathrm{jvp}(\bar{v},(z,r,t),(v,0,1))$. Mean Flow uses this identity to train an average-velocity predictor for one-step generation~\citep{geng2025mean}. The iMF motivation in our source notes is that a nonlinear JVP expression should use marginal rather than conditional velocity; directly substituting conditional velocity can make the objective unstable. iMF instead uses a corrected relation
\begin{equation}
V_\theta(z_t) = u_\theta(z_t) + (t-r)\mathrm{JVP}_{\mathrm{sg}}(u_\theta; v_\theta),
\end{equation}
where gradients update the average-flow branch $u_\theta$ while the JVP direction is stop-gradient. This preserves the practical stability of flow-matching-style supervision while training the model used for fast inference~\citep{geng2025improved}.
\paragraph{Residual aggregation and attention residuals.}
Standard residual networks update $x_t=x_{t-1}+y_t$, so recursively $x_t=y_0+y_1+\cdots+y_t$. Equivalently, the next layer receives a uniform aggregation of all previous residual increments. A natural generalization is
\begin{equation}
y_{t+1}=f_{t+1}\left(\sum_{s=0}^t a_{t+1,s}y_s\right),\qquad a_{t+1,s}\ge0,\quad \sum_{s=0}^t a_{t+1,s}=1.
\end{equation}
AttnRes instantiates the weights with an attention distribution over previous residual outputs,
\begin{equation}
a_{t+1,s}\propto\exp\left(w_{t+1}^{\top}\mathrm{RMSNorm}(y_s)\right).
\end{equation}
This preserves the residual pathway while allowing each layer to select useful earlier representations instead of uniformly accumulating all of them~\citep{kimi2026attention}. We use this mechanism in the policy transformer; our ablations suggest that applying AttnRes too broadly inside the vision stack is not automatically beneficial.
\section{Method}
\label{sec:method}
\subsection{Policy formulation}
\label{sec:policy_formulation}
We consider simulated imitation learning in RoboIMI. At each control step, the policy receives visual observations, robot state, and a task context, and predicts an action chunk of length \texttt{exec}. The main iMF-AttnRes socket and object-transfer policies use ph32/exec16 or related horizon-execution settings. The rollout evaluator records cumulative reward, median reward, mean maximum reward, nonzero-reward episode count, task-specific success-like threshold count, inference FPS, control FPS, and inference latency.
The policy contains three conceptual parts. First, an observation encoder maps visual and state inputs into policy tokens. Second, a transformer policy core uses AttnRes in place of fixed residual accumulation. Third, the action generator uses iMF to predict an average flow over action noise-to-data paths, enabling one-to-few inference evaluations. In the strongest socket variants, the generator uses infer2 or infer3; in the strongest object-transfer run, it uses infer1.
\subsection{Improved Mean Flow action generation}
\label{sec:imf_action_generation}
Classical flow matching trains an instantaneous velocity field. For action generation, a low number of integration steps can cause large path errors if the learned field is curved. Mean Flow addresses this by training an average velocity over a time interval. The model receives $(z_t,r,t)$ and predicts $u_\theta(z_t,r,t)$, an approximation to $\bar{v}(z_t,r,t)$. At inference, this average velocity is used to move directly across a larger interval, replacing a long sequence of small denoising or ODE steps.
The iMF training objective used in our implementation follows the corrected relation in the source notes:
\begin{equation}
V_\theta(z_t,r,t) = u_\theta(z_t,r,t) + (t-r)\mathrm{JVP}_{\mathrm{sg}}\big(u_\theta; v_\theta\big).
\end{equation}
Here $v_\theta$ is obtained from the same network at the instantaneous case, and the stop-gradient JVP direction prevents the nonlinear term from destabilizing the target. The supervised target remains the conditional straight-path velocity available from the imitation action sample and noise sample, but the learned branch used at deployment is the average-flow predictor $u_\theta$. This is the mechanism that allows infer1, infer2, and infer3 policies to compete with infer100 diffusion-style baselines.
\subsection{Attention Residual policy transformer}
\label{sec:attnres_policy}
In a standard PreNorm transformer, residual updates are added with fixed unit weights. From the residual-increment view, this means layer $t+1$ consumes a uniform sum of $y_0,\ldots,y_t$. We replace this fixed accumulation in the policy transformer with AttnRes. Each layer has a learned query vector $w_{t+1}$; previous residual increments are normalized and scored; the layer input is the weighted sum. This adds a lightweight depth-selection mechanism:
\begin{equation}
\tilde{x}_{t+1}=\sum_{s=0}^{t} \mathrm{softmax}_s\left(w_{t+1}^{\top}\mathrm{RMSNorm}(y_s)\right)y_s.
\end{equation}
The transformer block then computes $y_{t+1}=f_{t+1}(\tilde{x}_{t+1})$. The design is intended to reduce feature dilution in deep policies. In our object-transfer ablations, the best settings apply AttnRes in the policy transformer; replacing residuals throughout the vision encoder underperforms, suggesting that the placement of learned residual aggregation is important.
\subsection{Inference and control loop}
\label{sec:inference_loop}
At deployment, the iMF-AttnRes policy samples an action chunk with one to a few average-flow evaluations and then executes the chunk for \texttt{exec} control steps. This contrasts with diffusion-style baselines whose run names include infer100. For socket insertion, the strongest iMF-AttnRes policies use ph32/exec16 and infer2 or infer3. For object transfer, the strongest iMF-AttnRes run uses ph32/exec16 and infer1. The resulting control loop is simple: encode observations, run the AttnRes transformer, evaluate the iMF action generator a small number of times, execute the predicted chunk, and replan on the next observation window.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-method-overview}}]
\small Paper Banana prompt: create a clean 16:9 academic system diagram for iMF-AttnRes VLA. Show multi-view images, robot state, and optional language/task conditioning entering a VLA encoder; a policy transformer with AttnRes depth-wise residual aggregation; an improved Mean Flow action generator predicting average flow with one-to-few evaluations; and an exec-length action chunk controlling RoboIMI socket-insert and sim\_transfer robots. Use muted conference-paper colors, clear arrows, no decorative 3D, and labels for iMF, AttnRes, infer1/infer2/infer3, and exec16.
\end{tcolorbox}
\caption{Planned method overview figure. The current draft intentionally stores the Paper Banana generation prompt as a placeholder instead of generating an image.}
\label{fig:method_overview}
\end{figure}
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-imf-attnres-components}}]
\small Paper Banana prompt: create a 16:9 technical diagram contrasting classical flow matching instantaneous velocity $dx_t/dt=v(x_t,t)$; Mean Flow average velocity over $[r,t]$ with the JVP correction term; improved Mean Flow training relation $V_\theta(z_t)=u_\theta(z_t)+(t-r)\mathrm{JVP}_{\mathrm{sg}}(u_\theta;v_\theta)$; and AttnRes weighted residual aggregation $a_{t+1,s}$ proportional to $\exp(w_{t+1}\cdot\mathrm{RMSNorm}(y_s))$. Use equation callouts, minimal arrows, and a final callout saying one-to-few-step action generation for robot control.
\end{tcolorbox}
\caption{Planned technical component figure. The prompt is retained as a placeholder for later Paper Banana rendering.}
\label{fig:imf_attnres_components}
\end{figure}
\section{Experiments}
\label{sec:experiments}
\subsection{Setup and metrics}
\label{sec:setup_metrics}
We evaluate in two RoboIMI simulation environments. The socket peg task measures progress toward inserting a peg-like object into a socket. The sim\_transfer task measures object-transfer manipulation. Each reported row uses 100 rollouts. Metrics are copied directly from the rollout logs: average cumulative reward, median reward, mean maximum reward, maximum cumulative reward, count of nonzero-reward episodes, count of success-like episodes, inference FPS, control FPS, and average inference time when available. For socket insertion, the recorded success-like threshold is \texttt{max\_reward > 4}; for sim\_transfer, it is \texttt{max\_reward >= 4}. Hardware differs across runs, so speed comparisons are practical deployment measurements rather than perfectly normalized throughput benchmarks.
\subsection{Socket peg insertion}
\label{sec:socket_results}
\Cref{tab:socket_results} summarizes the socket peg insertion results. The best iMF-AttnRes policy by average reward is the infer3 model, with average reward 1513.56 and median reward 1901.5. It exceeds both diffusion-style infer100 baselines: the ph16 baseline obtains average reward 861.53 and median reward 668.0, while the ph32/exec16 baseline obtains average reward 1019.39 and median reward 804.0. The latency gap is large. The iMF-AttnRes infer2 and infer3 policies require 14.409 ms and 15.122 ms per inference, while the two infer100 baselines require 338.291 ms and 397.912 ms.
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{3.5pt}
\caption{Socket peg insertion over 100 rollouts. Success-like uses the recorded \texttt{max\_reward > 4} count.}
\label{tab:socket_results}
\begin{tabular}{llrrrrrr}
\toprule
Method & Setting & Avg. reward & Median & Avg. max & Nonzero & Success-like & Latency ms \\
\midrule
iMF-AttnRes & infer1, ph32, exec16, 50k & 1275.22 & 1490.5 & 3.12 & 84/100 & 0/100 & 16.186 \\
Diffusion-style VLA & infer100, ph16, 150k & 861.53 & 668.0 & 2.58 & 97/100 & 2/100 & 338.291 \\
Diffusion-style VLA & infer100, ph32, exec16, 150k & 1019.39 & 804.0 & 2.47 & 92/100 & 4/100 & 397.912 \\
iMF-AttnRes & infer2, ph32, exec16, 150k & 1472.62 & 1825.0 & 3.29 & 83/100 & 5/100 & 14.409 \\
iMF-AttnRes & infer3, ph32, exec16, 150k & \textbf{1513.56} & \textbf{1901.5} & 3.28 & 83/100 & 3/100 & 15.122 \\
ACT & action chunking & 289.60 & 13.0 & 1.29 & 55/100 & 1/100 & 100.212 \\
SmolVLA & 100k & 466.16 & 106.0 & 1.72 & 89/100 & 0/100 & \textbf{2.614} \\
\bottomrule
\end{tabular}
\end{table*}
The nonzero-reward count reveals an important caveat. The ph16 diffusion-style baseline has 97/100 nonzero episodes, higher than the iMF-AttnRes infer2 and infer3 policies at 83/100. Yet its average and median rewards are much lower. Thus, in this task, nonzero contact or partial progress is not sufficient; the iMF-AttnRes policies more often produce high-reward trajectories when they engage the task successfully. SmolVLA is the fastest method in raw latency and control FPS, but its reward is lower than iMF-AttnRes. ACT underperforms both iMF-AttnRes and the diffusion-style VLA baselines in this setup.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-socket-reward-latency}}]
\small Paper Banana prompt: create a 4:3 publication-quality reward-latency comparison for socket peg insertion. Plot methods from the socket table with average reward on the y-axis and average inference time or inverse latency on the x-axis. Highlight iMF-AttnRes infer2/infer3 at 1472.62/1513.56 average reward and 14.409/15.122 ms, diffusion-style infer100 baselines at 861.53/1019.39 average reward and 338.291/397.912 ms, ACT at 289.6 and 100.2116 ms, and SmolVLA at 466.16 and 2.614 ms. Use log-scale latency if helpful and label the speed-quality Pareto frontier.
\end{tcolorbox}
\caption{Planned socket reward-latency figure. The current draft contains the Paper Banana prompt placeholder instead of a rendered image.}
\label{fig:socket_reward_latency}
\end{figure}
\subsection{Object transfer and ablations}
\label{sec:sim_transfer_results}
\Cref{tab:sim_transfer_results} reports the object-transfer results. The strongest iMF-AttnRes run reaches average reward 526.22 and 44/100 success-like episodes, exceeding the native diffusion-policy DiT/DDPM/ResNet baseline at average reward 319.2 and 29/100 success-like episodes. This is the clearest sim\_transfer evidence that iMF-AttnRes can improve both reward and success-like threshold count.
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{3.2pt}
\caption{Object transfer / sim\_transfer over 100 rollouts. Success-like uses the recorded \texttt{max\_reward >= 4} count.}
\label{tab:sim_transfer_results}
\begin{tabular}{llrrrrrr}
\toprule
Method & Setting & Avg. reward & Median & Avg. max & Nonzero & Success-like & Inf. FPS \\
\midrule
Native Diffusion Policy & DiT + DDPM + ResNet & 319.20 & 6.0 & 1.68 & 55/100 & 29/100 & 32.09 \\
Diffusion-style VLA & emb384, layer18 & 233.52 & 0.0 & 1.12 & 39/100 & 17/100 & 1.859 \\
iMF multi-token ResNet18 & step34999, ph16, exec08 & 260.66 & 0.0 & 1.12 & 33/100 & 23/100 & \textbf{441.80} \\
iMF full AttnRes vision & ph16, exec08, 50k & 228.42 & 0.0 & 0.94 & 31/100 & 16/100 & 55.997 \\
iMF-AttnRes DiT only & ph16, exec16, 50k & 240.64 & 0.0 & 1.02 & 32/100 & 19/100 & 137.996 \\
iMF-AttnRes DiT only & ph32, exec08, 50k & 163.28 & 0.0 & 0.76 & 26/100 & 12/100 & 85.638 \\
iMF-AttnRes DiT only & ph32, exec32, 50k & 260.72 & 0.0 & 1.18 & 38/100 & 21/100 & 9.557 \\
iMF-AttnRes DiT only & ph16, exec08, 50k & 229.56 & 0.0 & 1.18 & 41/100 & 18/100 & 69.984 \\
iMF-AttnRes DiT only & ph08, exec08, 50k & 237.02 & 0.0 & 1.32 & 48/100 & 18/100 & 69.192 \\
iMF-AttnRes DiT only & ph32, exec16, 50k & 49.88 & 0.0 & 0.32 & 13/100 & 4/100 & 138.903 \\
iMF-AttnRes & infer1, ph32, exec16, 50k & \textbf{526.22} & \textbf{86.0} & \textbf{2.14} & \textbf{63/100} & \textbf{44/100} & 8.911 \\
\bottomrule
\end{tabular}
\end{table*}
The ablations are mixed and therefore useful. The ResNet18 multi-token iMF model is extremely fast at 441.80 inference FPS, but its average reward is 260.66, below the native diffusion-policy baseline. The full-AttnRes vision model reaches 228.42 average reward, suggesting that replacing residuals inside the vision encoder is not automatically helpful. Horizon and execution length are also sensitive: the ph32/exec16 DiT-only iMF variant reaches only 49.88 average reward despite high inference FPS. The strongest object-transfer run therefore combines iMF-AttnRes with the right horizon and execution setting rather than showing a universally dominant architectural change.
\begin{figure}[t]
\centering
\begin{tcolorbox}[width=0.96\linewidth,colback=white,colframe=black!35,title={Placeholder for \texttt{fig-sim-transfer-ablation}}]
\small Paper Banana prompt: create a 4:3 grouped bar chart for sim\_transfer ablations. Show average reward and success-like episode count for native diffusion policy, best sim-transfer iMF-AttnRes infer1, ResNet18 multi-token iMF, full-AttnRes vision, and selected horizon/execution variants. Emphasize that best iMF-AttnRes reaches 526.22 average reward and 44/100 success-like episodes versus native diffusion policy at 319.2 and 29/100, while several ablations underperform.
\end{tcolorbox}
\caption{Planned sim\_transfer ablation figure. The prompt is retained for later Paper Banana rendering.}
\label{fig:sim_transfer_ablation}
\end{figure}
\subsection{Derived speed-quality comparisons}
\label{sec:derived_comparisons}
\Cref{tab:derived_comparisons} lists the derived comparisons used to summarize the tradeoff. On socket insertion, iMF-AttnRes infer3 improves average reward by 1.76$\times$ over the ph16 diffusion-style baseline and by 1.48$\times$ over the ph32 diffusion-style baseline. The same infer3 run improves inference FPS by 22.97$\times$ over the ph16 diffusion-style baseline. The infer2 run is 23.48$\times$ lower latency than the ph16 diffusion-style baseline and 28.45$\times$ higher inference FPS than the ph32 diffusion-style baseline. On object transfer, best iMF-AttnRes improves average reward by 1.65$\times$ and success-like episodes by +15 over native Diffusion Policy.
\begin{table}[t]
\centering
\small
\caption{Derived comparisons from the 100-rollout logs.}
\label{tab:derived_comparisons}
\begin{tabular}{llr}
\toprule
Comparison & Metric & Result \\
\midrule
Socket iMF infer3 vs ph16 diffusion & Avg. reward & 1.76$\times$ \\
Socket iMF infer3 vs ph32 diffusion & Avg. reward & 1.48$\times$ \\
Socket iMF infer3 vs ACT & Avg. reward & 5.23$\times$ \\
Socket iMF infer3 vs SmolVLA & Avg. reward & 3.25$\times$ \\
Socket iMF infer2 vs ph16 diffusion & Latency reduction & 23.48$\times$ \\
Socket iMF infer3 vs ph16 diffusion & Inference FPS & 22.97$\times$ \\
Socket iMF infer2 vs ph32 diffusion & Inference FPS & 28.45$\times$ \\
Sim transfer iMF vs native diffusion & Avg. reward & 1.65$\times$ \\
Sim transfer iMF vs native diffusion & Success-like episodes & +15 \\
ResNet18 iMF vs layer18 baseline & Inference FPS & 237.65$\times$ \\
\bottomrule
\end{tabular}
\end{table}
\section{Limitations}
\label{sec:limitations}
All experiments are simulation rollouts. The data do not establish real-robot transfer, robustness to sensing changes, or safety under hardware execution. CoRL submissions should provide convincing robotics evidence; the present draft should therefore be viewed as a simulation-first manuscript that still needs real-robot or stronger transfer evidence before formal submission.
The speed measurements are also not perfectly hardware-normalized. Runs were collected on RTX 5880 Ada, L20, and RTX 5090 machines as encoded by their run names. Large latency differences between infer100 diffusion-style policies and infer1--3 iMF-AttnRes policies are meaningful because they reflect algorithmic sampling cost, but exact FPS ratios may include hardware and implementation effects. Future experiments should rerun all main policies on the same GPU, with identical rollout parallelism and identical observation preprocessing.
Finally, iMF-AttnRes is sensitive to design choices. In sim\_transfer, several iMF variants underperform the native diffusion-policy baseline, and full AttnRes in the vision encoder performs worse than more conservative policy-transformer AttnRes. This suggests that average-flow training is not a plug-in guarantee; horizon, execution length, visual tokenization, and residual placement must be tuned carefully.
\section{Conclusion}
\label{sec:conclusion}
We presented a first simulation study of iMF-AttnRes for fast robot action generation. The method adapts improved Mean Flow to VLA imitation learning and uses Attention Residuals to replace fixed residual accumulation in the policy transformer. In socket peg insertion, iMF-AttnRes achieves higher average and median reward than diffusion-style infer100 baselines while reducing inference latency from hundreds of milliseconds to roughly 15 ms. In object transfer, the best iMF-AttnRes run improves average reward and success-like episode count over the native diffusion-policy baseline. At the same time, ablations show that the method is sensitive to horizon/execution settings and residual placement. The next step is controlled, hardware-normalized evaluation with real-robot transfer.
\clearpage
\bibliography{refs}
\end{document}
+7
View File
@@ -0,0 +1,7 @@
[
{
"iteration": 0,
"decision": "accepted_first_draft",
"notes": "Generated first version only. No iterative content-refinement loop was run because user asked to first write one version for inspection."
}
]
+140
View File
@@ -0,0 +1,140 @@
@article{chi2023diffusion,
title = {Diffusion Policy: Visuomotor Policy Learning via Action Diffusion},
author = {Chi, Cheng and Xu, Zhenjia and Feng, Siyuan and Cousineau, Eric and Du, Yilun and Burchfiel, Benjamin and Tedrake, Russ and Song, Shuran},
year = {2023},
journal = {arXiv preprint arXiv:2303.04137},
eprint = {2303.04137},
archivePrefix = {arXiv}
}
@inproceedings{ho2020denoising,
title = {Denoising Diffusion Probabilistic Models},
author = {Ho, Jonathan and Jain, Ajay and Abbeel, Pieter},
year = {2020},
booktitle = {Advances in Neural Information Processing Systems}
}
@inproceedings{peebles2023scalable,
title = {Scalable Diffusion Models with Transformers},
author = {Peebles, William and Xie, Saining},
year = {2023},
booktitle = {IEEE/CVF International Conference on Computer Vision}
}
@inproceedings{lipman2023flow,
title = {Flow Matching for Generative Modeling},
author = {Lipman, Yaron and Chen, Ricky T. Q. and Ben-Hamu, Heli and Nickel, Maximilian and Le, Matt},
year = {2023},
booktitle = {International Conference on Learning Representations}
}
@article{liu2022flow,
title = {Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow},
author = {Liu, Xingchao and Gong, Chengyue and Liu, Qiang},
year = {2022},
journal = {arXiv preprint arXiv:2209.03003},
eprint = {2209.03003},
archivePrefix = {arXiv}
}
@article{geng2025mean,
title = {Mean Flows for One-step Generative Modeling},
author = {Geng, Zhengyang and Deng, Mingyang and Bai, Xingjian and Kolter, J. Zico and He, Kaiming},
year = {2025},
journal = {arXiv preprint arXiv:2505.13447},
eprint = {2505.13447},
archivePrefix = {arXiv}
}
@article{geng2025improved,
title = {Improved Mean Flows: On the Challenges of Fastforward Generative Models},
author = {Geng, Zhengyang and Lu, Yiyang and Wu, Zongze and Shechtman, Eli and Kolter, J. Zico and He, Kaiming},
year = {2025},
journal = {arXiv preprint arXiv:2512.02012},
eprint = {2512.02012},
archivePrefix = {arXiv}
}
@article{brohan2022rt,
title = {RT-1: Robotics Transformer for Real-World Control at Scale},
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Dabis, Joseph and Finn, Chelsea and Gopalakrishnan, Keerthana and Hausman, Karol and Herzog, Alexander and Hsu, Jasmine and others},
year = {2022},
journal = {arXiv preprint arXiv:2212.06817},
eprint = {2212.06817},
archivePrefix = {arXiv}
}
@inproceedings{brohan2023rt,
title = {RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control},
author = {Brohan, Anthony and Brown, Noah and Carbajal, Justice and Chebotar, Yevgen and Chen, Xi and Choromanski, Krzysztof and Ding, Tianli and Driess, Danny and Dubey, Avinava and Finn, Chelsea and others},
year = {2023},
booktitle = {Conference on Robot Learning}
}
@article{kim2024openvla,
title = {OpenVLA: An Open-Source Vision-Language-Action Model},
author = {Kim, Moo Jin and Pertsch, Karl and Karamcheti, Siddharth and Xiao, Ted and Balakrishna, Ashwin and Nair, Suraj and Rafailov, Rafael and Foster, Ethan and Lam, Grace and Sanketi, Pannag and others},
year = {2024},
journal = {arXiv preprint arXiv:2406.09246},
eprint = {2406.09246},
archivePrefix = {arXiv}
}
@article{shukor2025smolvla,
title = {SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},
author = {Shukor, Mustafa and Aubakirova, Dana and Capuano, Francesco and Kooijmans, Pepijn and Palma, Steven and Zouitine, Adil and Aractingi, Michel and Pascal, Caroline and Russi, Martino and Marafioti, Andres and others},
year = {2025},
journal = {arXiv preprint arXiv:2506.01844},
eprint = {2506.01844},
archivePrefix = {arXiv}
}
@article{black2024pi0,
title = {{$\pi_0$}: A Vision-Language-Action Flow Model for General Robot Control},
author = {Black, Kevin and Brown, Noah and Driess, Danny and Esmail, Adnan and Equi, Michael and Finn, Chelsea and Fusai, Niccolo and Groom, Lachy and Hausman, Karol and Ichter, Brian and others},
year = {2024},
journal = {arXiv preprint arXiv:2410.24164},
eprint = {2410.24164},
archivePrefix = {arXiv}
}
@article{pertsch2025fast,
title = {FAST: Efficient Action Tokenization for Vision-Language-Action Models},
author = {Pertsch, Karl and Stachowicz, Kyle and Ichter, Brian and Driess, Danny and Nair, Suraj and Vuong, Quan and Mees, Oier and Finn, Chelsea and Levine, Sergey},
year = {2025},
journal = {arXiv preprint arXiv:2501.09747},
eprint = {2501.09747},
archivePrefix = {arXiv}
}
@article{zhao2023learning,
title = {Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware},
author = {Zhao, Tony Z. and Kumar, Vikash and Levine, Sergey and Finn, Chelsea},
year = {2023},
journal = {arXiv preprint arXiv:2304.13705},
eprint = {2304.13705},
archivePrefix = {arXiv}
}
@inproceedings{he2016deep,
title = {Deep Residual Learning for Image Recognition},
author = {He, Kaiming and Zhang, Xiangyu and Ren, Shaoqing and Sun, Jian},
year = {2016},
booktitle = {IEEE Conference on Computer Vision and Pattern Recognition}
}
@inproceedings{vaswani2017attention,
title = {Attention Is All You Need},
author = {Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N. and Kaiser, Lukasz and Polosukhin, Illia},
year = {2017},
booktitle = {Advances in Neural Information Processing Systems}
}
@article{kimi2026attention,
title = {Attention Residuals},
author = {{Kimi Team}},
year = {2026},
journal = {arXiv preprint arXiv:2603.15031},
eprint = {2603.15031},
archivePrefix = {arXiv}
}
+21
View File
@@ -0,0 +1,21 @@
{
"available": [
"cleveref",
"nicefrac",
"microtype",
"url",
"booktabs",
"natbib",
"hyperref",
"fontenc",
"times",
"lmodern"
],
"missing": [],
"use_cleveref": true,
"use_nicefrac": true,
"use_microtype": true,
"use_t1_fontenc": true,
"tex_binary": "pdflatex",
"checked_at": "2026-05-15T11:10:11.905693"
}