{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:IIV2NAGY4RUP6YJKIV36VQLL6H","short_pith_number":"pith:IIV2NAGY","schema_version":"1.0","canonical_sha256":"422ba680d8e468ff612a4577eac16bf1de91f6a158d89bd562bf58d258703dfe","source":{"kind":"arxiv","id":"2106.09203","version":1},"attestation_state":"computed","paper":{"title":"Learning from Demonstration without Demonstrations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Gilad Francis, Philippe Morere, Tom Blau","submitted_at":"2021-06-17T01:57:08Z","abstract_excerpt":"State-of-the-art reinforcement learning (RL) algorithms suffer from high sample complexity, particularly in the sparse reward case. A popular strategy for mitigating this problem is to learn control policies by imitating a set of expert demonstrations. The drawback of such approaches is that an expert needs to produce demonstrations, which may be costly in practice. To address this shortcoming, we propose Probabilistic Planning for Demonstration Discovery (P2D2), a technique for automatically discovering demonstrations without access to an expert. We formulate discovering demonstrations as a s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.09203","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-17T01:57:08Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"27282781110036b4bb8716a20b39ac41e7f7d9586ae221ad5a009a9b9dfdd707","abstract_canon_sha256":"3d20bf72ac06d4540a1955719d1fe89f68aa24a5b894f352374dfd6ad758aaf2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:13.420858Z","signature_b64":"ncyrWteWXppjsg4+YKDSkF+kvUI9oFKpX3to9hidrFsnieBxcahlquw7Q+HL8SmHAxoME5A/QpQIVfjQQAbPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"422ba680d8e468ff612a4577eac16bf1de91f6a158d89bd562bf58d258703dfe","last_reissued_at":"2026-07-05T02:50:13.419717Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:13.419717Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Demonstration without Demonstrations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Gilad Francis, Philippe Morere, Tom Blau","submitted_at":"2021-06-17T01:57:08Z","abstract_excerpt":"State-of-the-art reinforcement learning (RL) algorithms suffer from high sample complexity, particularly in the sparse reward case. A popular strategy for mitigating this problem is to learn control policies by imitating a set of expert demonstrations. The drawback of such approaches is that an expert needs to produce demonstrations, which may be costly in practice. To address this shortcoming, we propose Probabilistic Planning for Demonstration Discovery (P2D2), a technique for automatically discovering demonstrations without access to an expert. We formulate discovering demonstrations as a s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09203","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.09203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.09203","created_at":"2026-07-05T02:50:13.419769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.09203v1","created_at":"2026-07-05T02:50:13.419769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09203","created_at":"2026-07-05T02:50:13.419769+00:00"},{"alias_kind":"pith_short_12","alias_value":"IIV2NAGY4RUP","created_at":"2026-07-05T02:50:13.419769+00:00"},{"alias_kind":"pith_short_16","alias_value":"IIV2NAGY4RUP6YJK","created_at":"2026-07-05T02:50:13.419769+00:00"},{"alias_kind":"pith_short_8","alias_value":"IIV2NAGY","created_at":"2026-07-05T02:50:13.419769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H","json":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H.json","graph_json":"https://pith.science/api/pith-number/IIV2NAGY4RUP6YJKIV36VQLL6H/graph.json","events_json":"https://pith.science/api/pith-number/IIV2NAGY4RUP6YJKIV36VQLL6H/events.json","paper":"https://pith.science/paper/IIV2NAGY"},"agent_actions":{"view_html":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H","download_json":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H.json","view_paper":"https://pith.science/paper/IIV2NAGY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.09203&json=true","fetch_graph":"https://pith.science/api/pith-number/IIV2NAGY4RUP6YJKIV36VQLL6H/graph.json","fetch_events":"https://pith.science/api/pith-number/IIV2NAGY4RUP6YJKIV36VQLL6H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H/action/storage_attestation","attest_author":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H/action/author_attestation","sign_citation":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H/action/citation_signature","submit_replication":"https://pith.science/pith/IIV2NAGY4RUP6YJKIV36VQLL6H/action/replication_record"}},"created_at":"2026-07-05T02:50:13.419769+00:00","updated_at":"2026-07-05T02:50:13.419769+00:00"}