{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YHWFIQJMSCJC6IQLU7FAFV6RYW","short_pith_number":"pith:YHWFIQJM","schema_version":"1.0","canonical_sha256":"c1ec54412c90922f220ba7ca02d7d1c5ab2c46c5b48e4b90e27c268dae5b7d59","source":{"kind":"arxiv","id":"2409.09491","version":2},"attestation_state":"computed","paper":{"title":"Robot Learning as an Empirical Science: Best Practices for Policy Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Benjamin Burchfiel, Gordon Richardson, Hadas Kress-Gazit, Kunimatsu Hashimoto, Naveen Kuppuswamy, Paarth Shah, Phoebe Horgan, Siyuan Feng","submitted_at":"2024-09-14T17:38:04Z","abstract_excerpt":"The robot learning community has made great strides in recent years, proposing new architectures and showcasing impressive new capabilities; however, the dominant metric used in the literature, especially for physical experiments, is \"success rate\", i.e. the percentage of runs that were successful. Furthermore, it is common for papers to report this number with little to no information regarding the number of runs, the initial conditions, and the success criteria, little to no narrative description of the behaviors and failures observed, and little to no statistical analysis of the findings. I"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.09491","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-09-14T17:38:04Z","cross_cats_sorted":[],"title_canon_sha256":"85012eaf9f5b5a919103ac2a8427d20afccb770aa6d50d879e1d0db024da1984","abstract_canon_sha256":"2525bfd980f465e9f7a357673de7f0fb1a52c9cacd528d89a9a6ecfb7a8aa010"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:04.242625Z","signature_b64":"vD+gemcOhRWAq6/vJUk9vRtbk9hUvvDyMgi/fS7WfqqX9coiOXvmvJELpt1s20fsTIAQTU2NgUvVuN9SnsV5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c1ec54412c90922f220ba7ca02d7d1c5ab2c46c5b48e4b90e27c268dae5b7d59","last_reissued_at":"2026-07-05T09:10:04.241871Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:04.241871Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robot Learning as an Empirical Science: Best Practices for Policy Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Benjamin Burchfiel, Gordon Richardson, Hadas Kress-Gazit, Kunimatsu Hashimoto, Naveen Kuppuswamy, Paarth Shah, Phoebe Horgan, Siyuan Feng","submitted_at":"2024-09-14T17:38:04Z","abstract_excerpt":"The robot learning community has made great strides in recent years, proposing new architectures and showcasing impressive new capabilities; however, the dominant metric used in the literature, especially for physical experiments, is \"success rate\", i.e. the percentage of runs that were successful. Furthermore, it is common for papers to report this number with little to no information regarding the number of runs, the initial conditions, and the success criteria, little to no narrative description of the behaviors and failures observed, and little to no statistical analysis of the findings. I"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.09491","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.09491/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.09491","created_at":"2026-07-05T09:10:04.242121+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.09491v2","created_at":"2026-07-05T09:10:04.242121+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.09491","created_at":"2026-07-05T09:10:04.242121+00:00"},{"alias_kind":"pith_short_12","alias_value":"YHWFIQJMSCJC","created_at":"2026-07-05T09:10:04.242121+00:00"},{"alias_kind":"pith_short_16","alias_value":"YHWFIQJMSCJC6IQL","created_at":"2026-07-05T09:10:04.242121+00:00"},{"alias_kind":"pith_short_8","alias_value":"YHWFIQJM","created_at":"2026-07-05T09:10:04.242121+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21572","citing_title":"Robot Critics that Sweat the Small Stuff","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09813","citing_title":"iMaC: Translating Actions into Motion and Contact Images for Embodied World Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04130","citing_title":"CLAW: Learning Continuous Latent Action World Models via Adversarial Latent Regularization","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31494","citing_title":"Robustness of Robotic Manipulation: Foundations and Frontiers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29710","citing_title":"PhAIL: A Real-Robot VLA Benchmark and Distributional Methodology","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01036","citing_title":"Position: Good Embodied Reward Models Need Bad Behavior Data","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05331","citing_title":"A Careful Examination of Large Behavior Models for Multitask Dexterous Manipulation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23847","citing_title":"Instrumentation for Imitation Learning: Enhancing Training Datasets for Clothes Hanger Insertion","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01433","citing_title":"AFFORD2ACT: Affordance-Guided Automatic Keypoint Selection for Generalizable and Lightweight Robotic Manipulation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17321","citing_title":"Neuro-Symbolic Control with Large Language Models for Language-Guided Spatial Tasks","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09860","citing_title":"RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24018","citing_title":"Betting for Sim-to-Real Performance Evaluation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09860","citing_title":"RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW","json":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW.json","graph_json":"https://pith.science/api/pith-number/YHWFIQJMSCJC6IQLU7FAFV6RYW/graph.json","events_json":"https://pith.science/api/pith-number/YHWFIQJMSCJC6IQLU7FAFV6RYW/events.json","paper":"https://pith.science/paper/YHWFIQJM"},"agent_actions":{"view_html":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW","download_json":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW.json","view_paper":"https://pith.science/paper/YHWFIQJM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.09491&json=true","fetch_graph":"https://pith.science/api/pith-number/YHWFIQJMSCJC6IQLU7FAFV6RYW/graph.json","fetch_events":"https://pith.science/api/pith-number/YHWFIQJMSCJC6IQLU7FAFV6RYW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW/action/storage_attestation","attest_author":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW/action/author_attestation","sign_citation":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW/action/citation_signature","submit_replication":"https://pith.science/pith/YHWFIQJMSCJC6IQLU7FAFV6RYW/action/replication_record"}},"created_at":"2026-07-05T09:10:04.242121+00:00","updated_at":"2026-07-05T09:10:04.242121+00:00"}