{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JHUGV5MKLKBYYKDLAUPSNS73ES","short_pith_number":"pith:JHUGV5MK","schema_version":"1.0","canonical_sha256":"49e86af58a5a838c286b051f26cbfb24b0f3850f648d232e7dc28f30db16207e","source":{"kind":"arxiv","id":"2502.08623","version":3},"attestation_state":"computed","paper":{"title":"Robot Data Curation with Mutual Information Estimators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Annie Xie, Ashwin Balakrishna, Ayzaan Wahid, Coline Devin, Dhruv Shah, Dorsa Sadigh, Joey Hejna, Jonathan Tompson, Pannag Sanketi, Suvir Mirchandani","submitted_at":"2025-02-12T18:23:23Z","abstract_excerpt":"The performance of imitation learning policies often hinges on the datasets with which they are trained. Consequently, investment in data collection for robotics has grown across both industrial and academic labs. However, despite the marked increase in the quantity of demonstrations collected, little work has sought to assess the quality of said data despite mounting evidence of its importance in other areas such as vision and language. In this work, we take a critical step towards addressing the data quality in robotics. Given a dataset of demonstrations, we aim to estimate the relative qual"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08623","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-02-12T18:23:23Z","cross_cats_sorted":[],"title_canon_sha256":"cc35915982e05b2c623ae59ec96dfabe930c95e1588d2061dab67db29c32223b","abstract_canon_sha256":"c2d303e54de2bbcfc71856bac3944dc376ae466017fa737f9a1f7e01f7f14eae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:19.475061Z","signature_b64":"v9BL2bmTRXOZHWdG2+u1ovuHV9JmqopQYQ9EonpUeT+5weqk9ELcql1it1p8pThE3tS/tzXtleOLpsrxx7pAAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49e86af58a5a838c286b051f26cbfb24b0f3850f648d232e7dc28f30db16207e","last_reissued_at":"2026-07-05T10:52:19.474581Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:19.474581Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robot Data Curation with Mutual Information Estimators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Annie Xie, Ashwin Balakrishna, Ayzaan Wahid, Coline Devin, Dhruv Shah, Dorsa Sadigh, Joey Hejna, Jonathan Tompson, Pannag Sanketi, Suvir Mirchandani","submitted_at":"2025-02-12T18:23:23Z","abstract_excerpt":"The performance of imitation learning policies often hinges on the datasets with which they are trained. Consequently, investment in data collection for robotics has grown across both industrial and academic labs. However, despite the marked increase in the quantity of demonstrations collected, little work has sought to assess the quality of said data despite mounting evidence of its importance in other areas such as vision and language. In this work, we take a critical step towards addressing the data quality in robotics. Given a dataset of demonstrations, we aim to estimate the relative qual"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08623","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08623/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08623","created_at":"2026-07-05T10:52:19.474638+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08623v3","created_at":"2026-07-05T10:52:19.474638+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08623","created_at":"2026-07-05T10:52:19.474638+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHUGV5MKLKBY","created_at":"2026-07-05T10:52:19.474638+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHUGV5MKLKBYYKDL","created_at":"2026-07-05T10:52:19.474638+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHUGV5MK","created_at":"2026-07-05T10:52:19.474638+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06442","citing_title":"SIEVE: Structure-Aware Data Selection for Imitation Learning with VLA Models","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23371","citing_title":"TSD: A Physics-Inspired Trajectory Saliency Detector for Efficient Imitation Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19998","citing_title":"Tri-Info: Generalizable, Interpretable Failure Prediction for VLA Models via Information Theory","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12365","citing_title":"Ambient Diffusion Policy: Imitation Learning from Suboptimal Data in Robotics","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01051","citing_title":"AutoSpeed: Annotation-Free Stage-Adaptive Motion Speed Learning for Robot Manipulation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03188","citing_title":"GeoSem-WAM: Geometry- and Semantic-Aware World Action Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13548","citing_title":"AttenA+: Rectifying Action Inequality in Robotic Foundation Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13548","citing_title":"AttenA+: Rectifying Action Inequality in Robotic Foundation Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13757","citing_title":"FrameSkip: Learning from Fewer but More Informative Frames in VLA Training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01529","citing_title":"Good in Bad (GiB): Sifting Through End-user Demonstrations for Learning a Better Policy","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23000","citing_title":"Learning from the Best: Smoothness-Driven Metrics for Data Quality in Imitation Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01544","citing_title":"An Efficient Metric for Data Quality Measurement in Imitation Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01529","citing_title":"Good in Bad (GiB): Sifting Through End-user Demonstrations for Learning a Better Policy","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES","json":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES.json","graph_json":"https://pith.science/api/pith-number/JHUGV5MKLKBYYKDLAUPSNS73ES/graph.json","events_json":"https://pith.science/api/pith-number/JHUGV5MKLKBYYKDLAUPSNS73ES/events.json","paper":"https://pith.science/paper/JHUGV5MK"},"agent_actions":{"view_html":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES","download_json":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES.json","view_paper":"https://pith.science/paper/JHUGV5MK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08623&json=true","fetch_graph":"https://pith.science/api/pith-number/JHUGV5MKLKBYYKDLAUPSNS73ES/graph.json","fetch_events":"https://pith.science/api/pith-number/JHUGV5MKLKBYYKDLAUPSNS73ES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES/action/storage_attestation","attest_author":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES/action/author_attestation","sign_citation":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES/action/citation_signature","submit_replication":"https://pith.science/pith/JHUGV5MKLKBYYKDLAUPSNS73ES/action/replication_record"}},"created_at":"2026-07-05T10:52:19.474638+00:00","updated_at":"2026-07-05T10:52:19.474638+00:00"}