@inproceedings{bibcite_121, author = {Bowen Feng and Keqin Wang and Yulong Yang and Amogh Joshi and Christine Allen-Blanchette and Dhruv Shah and Felix Heide}, title = {Training Generalist Navigation Policies via Action Self-Distillation}, abstract = {

Language provides a natural interface for humans to specify and reason about goals, but training language-conditioned robotic policies and reasoning typically requires expensive manual annotations. We propose Action Self-Distillation (ASD), a method for learning language-conditioned reasoning for robotic navigation without human-annotated language or reasoning labels. ASD uses an open-source VLM, which contains broad world knowledge and reasoning capabilities, as both a teacher network and a student policy. During training, the self-teacher receives privileged trajectory feedback and infers the underlying navigation goal together with goal-oriented reasoning. The student is then trained to condition on the inferred navigation goal, predict future ego-frame trajectory, and provide a description of the scene semantics, plan, and reasoning, from onboard observations alone. Across 13 challenging indoor and outdoor real-world navigation tasks, our method improves success rate over the strongest baseline by 30\% in indoor settings, 30\% on multi-stage tasks, and achieves 95\% success rate on outdoor tasks, validating that privileged trajectory feedback can enable self-distillation for scalable generalist navigation policies without human annotations.

}, year = {2026}, journal = {Annual Conference on Robot Learning (CoRL)}, }