{"work":{"id":"407838d2-13c3-4725-a0bb-cf6d303da780","openalex_id":"https://openalex.org/W7084075649","doi":"10.48550/arxiv.2509.25358","arxiv_id":"2509.25358","raw_key":null,"title":"SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation","authors":null,"authors_text":"Chen, Qianzhong, Yu, Justin, Schwager, Mac, Abbeel, Pieter, Shentu, Yide, Wu, Philipp","year":2025,"venue":"cs.RO","abstract":"Large-scale robot learning has made progress on complex manipulation tasks, yet long horizon, contact rich problems, especially those involving deformable objects, remain challenging due to inconsistent demonstration quality. We propose a stage-aware, video-based reward modeling framework that jointly predicts task stage and fine-grained progress, using natural language subtask annotations to derive consistent labels across variable-length demonstrations. This avoids the brittleness of frame index based labeling and provides stable supervision even in tasks like T-shirt folding. Our reward model is robust to demonstration variability, generalizes to out-of-distribution scenarios, and improves downstream policy training. Building on it, we introduce Reward-Aligned Behavior Cloning (RA-BC), which filters and reweights demonstrations based on reward estimates. Experiments show that our method significantly outperforms baselines in both real-world rollouts and human validation. On T-shirt folding, we achieve 83% success from the flattened state and 67% from the crumpled state, compared to 8% and 0% with vanilla BC. Overall, our results highlight reward modeling as a scalable and annotation-efficient solution for long horizon robotic manipulation. Project website: https://qianzhong-chen.github.io/sarm.github.io/","external_url":"https://arxiv.org/abs/2509.25358","cited_by_count":0,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2509.25358","created_at":"2026-05-11T20:46:12.122352+00:00","updated_at":"2026-08-05T02:49:54.815029+00:00","title_quality_ok":true,"display_title":"SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation","render_title":"SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation"},"hub":{"state":{"work_id":"407838d2-13c3-4725-a0bb-cf6d303da780","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":14,"external_cited_by_count":null,"distinct_field_count":3,"first_pith_cited_at":"2026-01-11T21:00:58+00:00","last_pith_cited_at":"2026-07-06T17:59:35+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-07-10T23:54:56.332220+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":2},{"context_role":"baseline","n":1}],"polarity_counts":[{"context_polarity":"background","n":2},{"context_polarity":"baseline","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}