{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KZ5MBCV6QNQRQTMJFTWGR4JVTJ","short_pith_number":"pith:KZ5MBCV6","schema_version":"1.0","canonical_sha256":"567ac08abe8361184d892cec68f1359a5c553a6bdc3de867738af8dc71bbeb4e","source":{"kind":"arxiv","id":"2503.03707","version":2},"attestation_state":"computed","paper":{"title":"Curating Demonstrations using Online Experience","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Alec M. Lessing, Annie S. Chen, Chelsea Finn, Yuejiang Liu","submitted_at":"2025-03-05T17:58:16Z","abstract_excerpt":"Many robot demonstration datasets contain heterogeneous demonstrations of varying quality. This heterogeneity may benefit policy pre-training, but can hinder robot performance when used with a final imitation learning objective. In particular, some strategies in the data may be less reliable than others or may be underrepresented in the data, leading to poor performance when such strategies are sampled at test time. Moreover, such unreliable or underrepresented strategies can be difficult even for people to discern, and sifting through demonstration datasets is time-consuming and costly. On th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.03707","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-03-05T17:58:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"68760aac4e6ec023fb8665942ef1f14f982ebd488f4b79615eb27b8eb680ac99","abstract_canon_sha256":"eaf52d7985ee466d023552a2a33a0023b804aef701ff714ee8f84fd81bd72053"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:52.817081Z","signature_b64":"VVlPikvyDEpZD82dd0q53BfWrWea9WDmE8B8IogWTiqdC/P+QwbqJbWLvyE8fwWk4e+VeWMCEQpA/LlyJ2vuAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"567ac08abe8361184d892cec68f1359a5c553a6bdc3de867738af8dc71bbeb4e","last_reissued_at":"2026-07-05T11:40:52.816428Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:52.816428Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Curating Demonstrations using Online Experience","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Alec M. Lessing, Annie S. Chen, Chelsea Finn, Yuejiang Liu","submitted_at":"2025-03-05T17:58:16Z","abstract_excerpt":"Many robot demonstration datasets contain heterogeneous demonstrations of varying quality. This heterogeneity may benefit policy pre-training, but can hinder robot performance when used with a final imitation learning objective. In particular, some strategies in the data may be less reliable than others or may be underrepresented in the data, leading to poor performance when such strategies are sampled at test time. Moreover, such unreliable or underrepresented strategies can be difficult even for people to discern, and sifting through demonstration datasets is time-consuming and costly. On th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.03707","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.03707/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.03707","created_at":"2026-07-05T11:40:52.816505+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.03707v2","created_at":"2026-07-05T11:40:52.816505+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.03707","created_at":"2026-07-05T11:40:52.816505+00:00"},{"alias_kind":"pith_short_12","alias_value":"KZ5MBCV6QNQR","created_at":"2026-07-05T11:40:52.816505+00:00"},{"alias_kind":"pith_short_16","alias_value":"KZ5MBCV6QNQRQTMJ","created_at":"2026-07-05T11:40:52.816505+00:00"},{"alias_kind":"pith_short_8","alias_value":"KZ5MBCV6","created_at":"2026-07-05T11:40:52.816505+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06442","citing_title":"SIEVE: Structure-Aware Data Selection for Imitation Learning with VLA Models","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23371","citing_title":"TSD: A Physics-Inspired Trajectory Saliency Detector for Efficient Imitation Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20871","citing_title":"Geometric Entropy: When Trajectory Diversity Helps and Hurts in Imitation Learning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13497","citing_title":"SPARC: Reliable Spatial Annotations from Robot Demonstrations at Scale","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01544","citing_title":"An Efficient Metric for Data Quality Measurement in Imitation Learning","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ","json":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ.json","graph_json":"https://pith.science/api/pith-number/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/graph.json","events_json":"https://pith.science/api/pith-number/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/events.json","paper":"https://pith.science/paper/KZ5MBCV6"},"agent_actions":{"view_html":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ","download_json":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ.json","view_paper":"https://pith.science/paper/KZ5MBCV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.03707&json=true","fetch_graph":"https://pith.science/api/pith-number/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/graph.json","fetch_events":"https://pith.science/api/pith-number/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/action/storage_attestation","attest_author":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/action/author_attestation","sign_citation":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/action/citation_signature","submit_replication":"https://pith.science/pith/KZ5MBCV6QNQRQTMJFTWGR4JVTJ/action/replication_record"}},"created_at":"2026-07-05T11:40:52.816505+00:00","updated_at":"2026-07-05T11:40:52.816505+00:00"}