{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:QSVFPS6FBLYFIQ5W4K3WSA3JKV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"df88fd9967bea8e463806e3501ce441b1649eeefe91e918be4a9db03ae196258","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-27T10:29:31Z","title_canon_sha256":"fdb70c5c114be41d1ab6fa074dea78bcb8282462f44308c66900ad03b9749eeb"},"schema_version":"1.0","source":{"id":"1905.11108","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1905.11108","created_at":"2026-07-05T00:07:23Z"},{"alias_kind":"arxiv_version","alias_value":"1905.11108v3","created_at":"2026-07-05T00:07:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.11108","created_at":"2026-07-05T00:07:23Z"},{"alias_kind":"pith_short_12","alias_value":"QSVFPS6FBLYF","created_at":"2026-07-05T00:07:23Z"},{"alias_kind":"pith_short_16","alias_value":"QSVFPS6FBLYFIQ5W","created_at":"2026-07-05T00:07:23Z"},{"alias_kind":"pith_short_8","alias_value":"QSVFPS6F","created_at":"2026-07-05T00:07:23Z"}],"graph_snapshots":[{"event_id":"sha256:6a99dbca083e3724aa2055ff28abfaff9cfde568d96fd043d0608e0b9445d947","target":"graph","created_at":"2026-07-05T00:07:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1905.11108/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Learning to imitate expert behavior from demonstrations can be challenging, especially in environments with high-dimensional, continuous observations and unknown dynamics. Supervised learning methods based on behavioral cloning (BC) suffer from distribution shift: because the agent greedily imitates demonstrated actions, it can drift away from demonstrated states due to error accumulation. Recent methods based on reinforcement learning (RL), such as inverse RL and generative adversarial imitation learning (GAIL), overcome this issue by training an RL agent to match the demonstrations over a lo","authors_text":"Anca D. Dragan, Sergey Levine, Siddharth Reddy","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-27T10:29:31Z","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.11108","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f0c8f305840cd17739103a34e73a68f1c50e89c2ba42a3203459ea1d173ae584","target":"record","created_at":"2026-07-05T00:07:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"df88fd9967bea8e463806e3501ce441b1649eeefe91e918be4a9db03ae196258","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-27T10:29:31Z","title_canon_sha256":"fdb70c5c114be41d1ab6fa074dea78bcb8282462f44308c66900ad03b9749eeb"},"schema_version":"1.0","source":{"id":"1905.11108","kind":"arxiv","version":3}},"canonical_sha256":"84aa57cbc50af05443b6e2b7690369554c5d5e298a2aed894e469122eea75a43","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"84aa57cbc50af05443b6e2b7690369554c5d5e298a2aed894e469122eea75a43","first_computed_at":"2026-07-05T00:07:23.477323Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:07:23.477323Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oYuH9d3c7aUek8ldr5fogjny4DQrYt3o+l8Abq4BS8vUxVPlg7+QWj+GmxD1NKNBTgnNohK23Pog7KUndRcmCg==","signature_status":"signed_v1","signed_at":"2026-07-05T00:07:23.477801Z","signed_message":"canonical_sha256_bytes"},"source_id":"1905.11108","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f0c8f305840cd17739103a34e73a68f1c50e89c2ba42a3203459ea1d173ae584","sha256:6a99dbca083e3724aa2055ff28abfaff9cfde568d96fd043d0608e0b9445d947"],"state_sha256":"1b9674077ef17de45aca07dffa47ccfa9632d82b3b43f83f8588b42ab730e86c"}