{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:VFE7JYTC4ZGPLGVI2XXDDQ2HK6","short_pith_number":"pith:VFE7JYTC","schema_version":"1.0","canonical_sha256":"a949f4e262e64cf59aa8d5ee31c34757872912c241a8bf94d952b23d93a7d817","source":{"kind":"arxiv","id":"2602.20220","version":2},"attestation_state":"computed","paper":{"title":"What Matters for Simulation to Online Reinforcement Learning on Real Robots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Andreas Krause, Chenhao Li, Dhruva Tirumala, Markus Wulfmeier, Ren\\'e Zurbr\\\"ugg, Stelian Coros, Yarden As","submitted_at":"2026-02-23T10:34:15Z","abstract_excerpt":"We investigate what specific design choices enable successful online reinforcement learning (RL) on physical robots. Across 100 real-world training runs on three distinct robotic platforms, we systematically ablate algorithmic, systems, and experimental decisions that are typically left implicit in prior work. We find that some widely used defaults can be harmful, while a set of robust, readily adopted design choices within standard RL practice yield stable learning across tasks and hardware. These results provide the first large-sample empirical study of such design choices, enabling practiti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.20220","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2026-02-23T10:34:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fadaddfb9e81192b7efab0621cc851aceeb6c5633e9b895ebbe65c134c2e464a","abstract_canon_sha256":"2d4233d6a41f50d4c7ff485927de7244c5a7e8dcbeb4de7894877c8005ed9f46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T00:23:07.737254Z","signature_b64":"fmoYkOqJL+2ItJF5EoMe7f28dL0Z8ncEBR8OqxiAkjfIXLfDCeK8sHYNG6pT9M1ChrVBrsiyRDIbUZVwMBaYBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a949f4e262e64cf59aa8d5ee31c34757872912c241a8bf94d952b23d93a7d817","last_reissued_at":"2026-07-24T00:23:07.736306Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T00:23:07.736306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Matters for Simulation to Online Reinforcement Learning on Real Robots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Andreas Krause, Chenhao Li, Dhruva Tirumala, Markus Wulfmeier, Ren\\'e Zurbr\\\"ugg, Stelian Coros, Yarden As","submitted_at":"2026-02-23T10:34:15Z","abstract_excerpt":"We investigate what specific design choices enable successful online reinforcement learning (RL) on physical robots. Across 100 real-world training runs on three distinct robotic platforms, we systematically ablate algorithmic, systems, and experimental decisions that are typically left implicit in prior work. We find that some widely used defaults can be harmful, while a set of robust, readily adopted design choices within standard RL practice yield stable learning across tasks and hardware. These results provide the first large-sample empirical study of such design choices, enabling practiti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.20220","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.20220/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.20220","created_at":"2026-07-24T00:23:07.736753+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.20220v2","created_at":"2026-07-24T00:23:07.736753+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.20220","created_at":"2026-07-24T00:23:07.736753+00:00"},{"alias_kind":"pith_short_12","alias_value":"VFE7JYTC4ZGP","created_at":"2026-07-24T00:23:07.736753+00:00"},{"alias_kind":"pith_short_16","alias_value":"VFE7JYTC4ZGPLGVI","created_at":"2026-07-24T00:23:07.736753+00:00"},{"alias_kind":"pith_short_8","alias_value":"VFE7JYTC","created_at":"2026-07-24T00:23:07.736753+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":5,"sample":[{"citing_arxiv_id":"2606.25527","citing_title":"Beyond One-Size-Fits-All: Diagnosis-Driven Online Reinforcement Learning with Offline Priors","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2605.24975","citing_title":"Bridging the Gap: Enabling Soft Actor Critic for High Performance Legged Locomotion","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6","json":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6.json","graph_json":"https://pith.science/api/pith-number/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/graph.json","events_json":"https://pith.science/api/pith-number/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/events.json","paper":"https://pith.science/paper/VFE7JYTC"},"agent_actions":{"view_html":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6","download_json":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6.json","view_paper":"https://pith.science/paper/VFE7JYTC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.20220&json=true","fetch_graph":"https://pith.science/api/pith-number/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/graph.json","fetch_events":"https://pith.science/api/pith-number/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/action/storage_attestation","attest_author":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/action/author_attestation","sign_citation":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/action/citation_signature","submit_replication":"https://pith.science/pith/VFE7JYTC4ZGPLGVI2XXDDQ2HK6/action/replication_record"}},"created_at":"2026-07-24T00:23:07.736753+00:00","updated_at":"2026-07-24T00:23:07.736753+00:00"}