{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EXYZIGTHLY73ZJNRDDP27WFV76","short_pith_number":"pith:EXYZIGTH","schema_version":"1.0","canonical_sha256":"25f1941a675e3fbca5b118dfafd8b5fface45c5f5696ee2ee155ecb9796abf96","source":{"kind":"arxiv","id":"2407.16677","version":4},"attestation_state":"computed","paper":{"title":"From Imitation to Refinement -- Residual RL for Precise Assembly","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Anthony Simeonov, Idan Shenfeld, Lars Ankile, Marcel Torne, Pulkit Agrawal","submitted_at":"2024-07-23T17:44:54Z","abstract_excerpt":"Recent advances in Behavior Cloning (BC) have made it easy to teach robots new tasks. However, we find that the ease of teaching comes at the cost of unreliable performance that saturates with increasing data for tasks requiring precision. The performance saturation can be attributed to two critical factors: (a) distribution shift resulting from the use of offline data and (b) the lack of closed-loop corrective control caused by action chucking (predicting a set of future actions executed open-loop) critical for BC performance. Our key insight is that by predicting action chunks, BC policies f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.16677","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-07-23T17:44:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"89c89149e85d287fe58d3e47f96717dd132d92d3fad9e944cc7f8f9e80192318","abstract_canon_sha256":"4862d3430a1146e3f687ef734b200edfb4805fad6d55635f795f93b8555f0bfb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:14.729729Z","signature_b64":"4YEgjsUaMUlMb2c4v4nu35ISEXOfhZZJfxeKFdexohPc5AVrlc17+a9G/mcWzonqAhaT6qzy8hPXEssGK19LDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25f1941a675e3fbca5b118dfafd8b5fface45c5f5696ee2ee155ecb9796abf96","last_reissued_at":"2026-07-05T09:48:14.729254Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:14.729254Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Imitation to Refinement -- Residual RL for Precise Assembly","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Anthony Simeonov, Idan Shenfeld, Lars Ankile, Marcel Torne, Pulkit Agrawal","submitted_at":"2024-07-23T17:44:54Z","abstract_excerpt":"Recent advances in Behavior Cloning (BC) have made it easy to teach robots new tasks. However, we find that the ease of teaching comes at the cost of unreliable performance that saturates with increasing data for tasks requiring precision. The performance saturation can be attributed to two critical factors: (a) distribution shift resulting from the use of offline data and (b) the lack of closed-loop corrective control caused by action chucking (predicting a set of future actions executed open-loop) critical for BC performance. Our key insight is that by predicting action chunks, BC policies f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.16677","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.16677/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.16677","created_at":"2026-07-05T09:48:14.729308+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.16677v4","created_at":"2026-07-05T09:48:14.729308+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.16677","created_at":"2026-07-05T09:48:14.729308+00:00"},{"alias_kind":"pith_short_12","alias_value":"EXYZIGTHLY73","created_at":"2026-07-05T09:48:14.729308+00:00"},{"alias_kind":"pith_short_16","alias_value":"EXYZIGTHLY73ZJNR","created_at":"2026-07-05T09:48:14.729308+00:00"},{"alias_kind":"pith_short_8","alias_value":"EXYZIGTH","created_at":"2026-07-05T09:48:14.729308+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21406","citing_title":"Robot Self-Improvement via Human-Video Dynamics Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12604","citing_title":"EgoEngine: From Egocentric Human Videos to High-Fidelity Dexterous Robot Demonstrations","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11408","citing_title":"Dynamic Execution Horizon Prediction for Chunk-based Robot Policies","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04525","citing_title":"HDFlow: Hierarchical Diffusion-Flow Planning for Long-horizon Tasks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19924","citing_title":"RoHIL: Robust Human-in-the-Loop Robotic Reinforcement Learning Against Illumination Variations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07986","citing_title":"EXPO: Stable Reinforcement Learning with Expressive Policies","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04525","citing_title":"HDFlow: Hierarchical Diffusion-Flow Planning for Long-horizon Tasks","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76","json":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76.json","graph_json":"https://pith.science/api/pith-number/EXYZIGTHLY73ZJNRDDP27WFV76/graph.json","events_json":"https://pith.science/api/pith-number/EXYZIGTHLY73ZJNRDDP27WFV76/events.json","paper":"https://pith.science/paper/EXYZIGTH"},"agent_actions":{"view_html":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76","download_json":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76.json","view_paper":"https://pith.science/paper/EXYZIGTH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.16677&json=true","fetch_graph":"https://pith.science/api/pith-number/EXYZIGTHLY73ZJNRDDP27WFV76/graph.json","fetch_events":"https://pith.science/api/pith-number/EXYZIGTHLY73ZJNRDDP27WFV76/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76/action/storage_attestation","attest_author":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76/action/author_attestation","sign_citation":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76/action/citation_signature","submit_replication":"https://pith.science/pith/EXYZIGTHLY73ZJNRDDP27WFV76/action/replication_record"}},"created_at":"2026-07-05T09:48:14.729308+00:00","updated_at":"2026-07-05T09:48:14.729308+00:00"}