{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FLBQJ2I35IQDL7K46ZD6DHTY6Y","short_pith_number":"pith:FLBQJ2I3","schema_version":"1.0","canonical_sha256":"2ac304e91bea2035fd5cf647e19e78f604d35c2d6857e5f44a524a074f7ee8bb","source":{"kind":"arxiv","id":"2409.20568","version":1},"attestation_state":"computed","paper":{"title":"Continuously Improving Mobile Manipulation with Autonomous Real-World RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Bernadette Bucher, Deepak Pathak, Emmanuel Panov, Jiuguang Wang, Russell Mendonca","submitted_at":"2024-09-30T17:59:50Z","abstract_excerpt":"We present a fully autonomous real-world RL framework for mobile manipulation that can learn policies without extensive instrumentation or human supervision. This is enabled by 1) task-relevant autonomy, which guides exploration towards object interactions and prevents stagnation near goal states, 2) efficient policy learning by leveraging basic task knowledge in behavior priors, and 3) formulating generic rewards that combine human-interpretable semantic information with low-level, fine-grained observations. We demonstrate that our approach allows Spot robots to continually improve their perf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.20568","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-09-30T17:59:50Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"335505a8966f343b79c4ec024a53871c32b3eabf8eca41925e634ebd3793517d","abstract_canon_sha256":"2b45ec33a09db54eec0682e18abc943bcd689c41dec06f1e5ee33c70c7b0f644"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:45.395912Z","signature_b64":"X3Rd78RhzVWjLIZ55oVaJYlFX+1fzEF4cAkranUS7EmR/djWcLWB3SaIdDNAvUxvZsvKR092jU6P0Q0ESeipDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ac304e91bea2035fd5cf647e19e78f604d35c2d6857e5f44a524a074f7ee8bb","last_reissued_at":"2026-07-05T09:13:45.395443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:45.395443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Continuously Improving Mobile Manipulation with Autonomous Real-World RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Bernadette Bucher, Deepak Pathak, Emmanuel Panov, Jiuguang Wang, Russell Mendonca","submitted_at":"2024-09-30T17:59:50Z","abstract_excerpt":"We present a fully autonomous real-world RL framework for mobile manipulation that can learn policies without extensive instrumentation or human supervision. This is enabled by 1) task-relevant autonomy, which guides exploration towards object interactions and prevents stagnation near goal states, 2) efficient policy learning by leveraging basic task knowledge in behavior priors, and 3) formulating generic rewards that combine human-interpretable semantic information with low-level, fine-grained observations. We demonstrate that our approach allows Spot robots to continually improve their perf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.20568","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.20568/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.20568","created_at":"2026-07-05T09:13:45.395497+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.20568v1","created_at":"2026-07-05T09:13:45.395497+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.20568","created_at":"2026-07-05T09:13:45.395497+00:00"},{"alias_kind":"pith_short_12","alias_value":"FLBQJ2I35IQD","created_at":"2026-07-05T09:13:45.395497+00:00"},{"alias_kind":"pith_short_16","alias_value":"FLBQJ2I35IQDL7K4","created_at":"2026-07-05T09:13:45.395497+00:00"},{"alias_kind":"pith_short_8","alias_value":"FLBQJ2I3","created_at":"2026-07-05T09:13:45.395497+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24466","citing_title":"FT-WBC: Learning Fault-Tolerant Whole-Body Control for Legged Loco-Manipulation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24466","citing_title":"FT-WBC: Learning Fault-Tolerant Whole-Body Control for Legged Loco-Manipulation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09023","citing_title":"TwinRL: Digital Twin-Driven Reinforcement Learning for Real-World Robotic Manipulation","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y","json":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y.json","graph_json":"https://pith.science/api/pith-number/FLBQJ2I35IQDL7K46ZD6DHTY6Y/graph.json","events_json":"https://pith.science/api/pith-number/FLBQJ2I35IQDL7K46ZD6DHTY6Y/events.json","paper":"https://pith.science/paper/FLBQJ2I3"},"agent_actions":{"view_html":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y","download_json":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y.json","view_paper":"https://pith.science/paper/FLBQJ2I3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.20568&json=true","fetch_graph":"https://pith.science/api/pith-number/FLBQJ2I35IQDL7K46ZD6DHTY6Y/graph.json","fetch_events":"https://pith.science/api/pith-number/FLBQJ2I35IQDL7K46ZD6DHTY6Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y/action/storage_attestation","attest_author":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y/action/author_attestation","sign_citation":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y/action/citation_signature","submit_replication":"https://pith.science/pith/FLBQJ2I35IQDL7K46ZD6DHTY6Y/action/replication_record"}},"created_at":"2026-07-05T09:13:45.395497+00:00","updated_at":"2026-07-05T09:13:45.395497+00:00"}