{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5A3LG6JDWOPITIWBCFX6B2OK3L","short_pith_number":"pith:5A3LG6JD","schema_version":"1.0","canonical_sha256":"e836b37923b39e89a2c1116fe0e9cadaf7809220d8a593e6f65496c9797b750d","source":{"kind":"arxiv","id":"2503.24278","version":2},"attestation_state":"computed","paper":{"title":"AutoEval: Autonomous Evaluation of Generalist Robot Manipulation Policies in the Real World","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Karl Pertsch, Pranav Atreya, Sergey Levine, You Liang Tan, Zhiyuan Zhou","submitted_at":"2025-03-31T16:23:44Z","abstract_excerpt":"Scalable and reproducible policy evaluation has been a long-standing challenge in robot learning. Evaluations are critical to assess progress and build better policies, but evaluation in the real world, especially at a scale that would provide statistically reliable results, is costly in terms of human time and hard to obtain. Evaluation of increasingly generalist robot policies requires an increasingly diverse repertoire of evaluation environments, making the evaluation bottleneck even more pronounced. To make real-world evaluation of robotic policies more practical, we propose AutoEval, a sy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.24278","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-03-31T16:23:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c048c137a2c47d639fd1ac4a14cd030864980fdd1204346f7cf19b3f4f460a3c","abstract_canon_sha256":"2907f9b45a7c043a7c8f947aa85c6a6a4c4719480fa3d1d397a27c671df16176"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:47.045023Z","signature_b64":"8tV9BPsxm35UHePUCmTF7lCQeYJV4Xqf6Vf5pw76EJ2tn1eWu3Tq8z3qwbwbBgvVQpFqSXLCltF8mosNskTAAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e836b37923b39e89a2c1116fe0e9cadaf7809220d8a593e6f65496c9797b750d","last_reissued_at":"2026-07-05T10:43:47.044565Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:47.044565Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoEval: Autonomous Evaluation of Generalist Robot Manipulation Policies in the Real World","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Karl Pertsch, Pranav Atreya, Sergey Levine, You Liang Tan, Zhiyuan Zhou","submitted_at":"2025-03-31T16:23:44Z","abstract_excerpt":"Scalable and reproducible policy evaluation has been a long-standing challenge in robot learning. Evaluations are critical to assess progress and build better policies, but evaluation in the real world, especially at a scale that would provide statistically reliable results, is costly in terms of human time and hard to obtain. Evaluation of increasingly generalist robot policies requires an increasingly diverse repertoire of evaluation environments, making the evaluation bottleneck even more pronounced. To make real-world evaluation of robotic policies more practical, we propose AutoEval, a sy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.24278","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.24278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.24278","created_at":"2026-07-05T10:43:47.044623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.24278v2","created_at":"2026-07-05T10:43:47.044623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.24278","created_at":"2026-07-05T10:43:47.044623+00:00"},{"alias_kind":"pith_short_12","alias_value":"5A3LG6JDWOPI","created_at":"2026-07-05T10:43:47.044623+00:00"},{"alias_kind":"pith_short_16","alias_value":"5A3LG6JDWOPITIWB","created_at":"2026-07-05T10:43:47.044623+00:00"},{"alias_kind":"pith_short_8","alias_value":"5A3LG6JD","created_at":"2026-07-05T10:43:47.044623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10366","citing_title":"A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09813","citing_title":"iMaC: Translating Actions into Motion and Contact Images for Embodied World Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08278","citing_title":"SIMPLE: Simulation-Based Policy Learning and Evaluation for Humanoid Loco-manipulation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01060","citing_title":"RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29898","citing_title":"Critical Interval MSE: Toward Reliable Offline Validation for Robot Manipulation Policies","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23204","citing_title":"AutoResearch AI: Towards AI-Powered Research Automation for Scientific Discovery","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05331","citing_title":"A Careful Examination of Large Behavior Models for Multitask Dexterous Manipulation","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20774","citing_title":"VLA-REPLICA: A Low-Cost, Reproducible Benchmark for Real-World Evaluation of Vision-Language-Action Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05635","citing_title":"Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02115","citing_title":"Robometer: Scaling General-Purpose Robotic Reward Models via Trajectory Comparisons","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22152","citing_title":"dWorldEval: Scalable Robotic Policy Evaluation via Discrete Diffusion World Model","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21741","citing_title":"Hi-WM: Human-in-the-World-Model for Scalable Robot Post-Training","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16788","citing_title":"LongBench: Evaluating Robotic Manipulation Policies on Real-World Long-Horizon Tasks","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L","json":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L.json","graph_json":"https://pith.science/api/pith-number/5A3LG6JDWOPITIWBCFX6B2OK3L/graph.json","events_json":"https://pith.science/api/pith-number/5A3LG6JDWOPITIWBCFX6B2OK3L/events.json","paper":"https://pith.science/paper/5A3LG6JD"},"agent_actions":{"view_html":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L","download_json":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L.json","view_paper":"https://pith.science/paper/5A3LG6JD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.24278&json=true","fetch_graph":"https://pith.science/api/pith-number/5A3LG6JDWOPITIWBCFX6B2OK3L/graph.json","fetch_events":"https://pith.science/api/pith-number/5A3LG6JDWOPITIWBCFX6B2OK3L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L/action/storage_attestation","attest_author":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L/action/author_attestation","sign_citation":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L/action/citation_signature","submit_replication":"https://pith.science/pith/5A3LG6JDWOPITIWBCFX6B2OK3L/action/replication_record"}},"created_at":"2026-07-05T10:43:47.044623+00:00","updated_at":"2026-07-05T10:43:47.044623+00:00"}