{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:RHGU5DOB3MSEXSSTZLZW6RCMX3","short_pith_number":"pith:RHGU5DOB","canonical_record":{"source":{"id":"2110.14770","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-27T21:05:00Z","cross_cats_sorted":[],"title_canon_sha256":"dadd884a872c1c205d08b8a7761cefae748d483315cc5652e8aaa7e86c66accf","abstract_canon_sha256":"143f4fc7ece4b7e132094ec9392ea8852ffa7d2618f469a772a8865d9e92abe2"},"schema_version":"1.0"},"canonical_sha256":"89cd4e8dc1db244bca53caf36f444cbefd4acee407708cce84c84b756061268f","source":{"kind":"arxiv","id":"2110.14770","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2110.14770","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"arxiv_version","alias_value":"2110.14770v1","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.14770","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"pith_short_12","alias_value":"RHGU5DOB3MSE","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"pith_short_16","alias_value":"RHGU5DOB3MSEXSST","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"pith_short_8","alias_value":"RHGU5DOB","created_at":"2026-07-05T03:26:46Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:RHGU5DOB3MSEXSSTZLZW6RCMX3","target":"record","payload":{"canonical_record":{"source":{"id":"2110.14770","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-27T21:05:00Z","cross_cats_sorted":[],"title_canon_sha256":"dadd884a872c1c205d08b8a7761cefae748d483315cc5652e8aaa7e86c66accf","abstract_canon_sha256":"143f4fc7ece4b7e132094ec9392ea8852ffa7d2618f469a772a8865d9e92abe2"},"schema_version":"1.0"},"canonical_sha256":"89cd4e8dc1db244bca53caf36f444cbefd4acee407708cce84c84b756061268f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:26:46.944320Z","signature_b64":"aJECvN9pRV1460t/Whenj42w9TSOcylJttkVp1eymaUCe6yFKWMLA5qsv66eJcpEXRl9eq2q3g7B8Ju9qZjvAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89cd4e8dc1db244bca53caf36f444cbefd4acee407708cce84c84b756061268f","last_reissued_at":"2026-07-05T03:26:46.943864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:26:46.943864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2110.14770","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:26:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Vc/wkH1AeElV5JGcHT0zKTrmva9FYFN+iMJCE82hQ+0XhgPXFNEbToOCxYzzRdb2xAhNo15q0L+oDD2dgj96CQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T18:51:08.247686Z"},"content_sha256":"042a2a243c76722bc92e98cc050c13a58056e30e868363629a7704543b5a2339","schema_version":"1.0","event_id":"sha256:042a2a243c76722bc92e98cc050c13a58056e30e868363629a7704543b5a2339"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:RHGU5DOB3MSEXSSTZLZW6RCMX3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"TRAIL: Near-Optimal Imitation Learning with Suboptimal Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Mengjiao Yang, Ofir Nachum, Sergey Levine","submitted_at":"2021-10-27T21:05:00Z","abstract_excerpt":"The aim in imitation learning is to learn effective policies by utilizing near-optimal expert demonstrations. However, high-quality demonstrations from human experts can be expensive to obtain in large numbers. On the other hand, it is often much easier to obtain large quantities of suboptimal or task-agnostic trajectories, which are not useful for direct imitation, but can nevertheless provide insight into the dynamical structure of the environment, showing what could be done in the environment even if not what should be done. We ask the question, is it possible to utilize such suboptimal off"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.14770","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.14770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:26:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yMtoNvosOIy7iXaNRqx5u9xCvAMUrOrW8pKGPK5iC5G8IvrUoIX8R1jPWvrOMaFKnWTQfz0eVhJMix6G0e5aBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T18:51:08.248212Z"},"content_sha256":"9ae7c6bbc0e4a16fbc972f3ecd024aeda0435f9fecad46f6ab0e5e10548b4cf4","schema_version":"1.0","event_id":"sha256:9ae7c6bbc0e4a16fbc972f3ecd024aeda0435f9fecad46f6ab0e5e10548b4cf4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3/bundle.json","state_url":"https://pith.science/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T18:51:08Z","links":{"resolver":"https://pith.science/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3","bundle":"https://pith.science/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3/bundle.json","state":"https://pith.science/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RHGU5DOB3MSEXSSTZLZW6RCMX3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:RHGU5DOB3MSEXSSTZLZW6RCMX3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"143f4fc7ece4b7e132094ec9392ea8852ffa7d2618f469a772a8865d9e92abe2","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-27T21:05:00Z","title_canon_sha256":"dadd884a872c1c205d08b8a7761cefae748d483315cc5652e8aaa7e86c66accf"},"schema_version":"1.0","source":{"id":"2110.14770","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2110.14770","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"arxiv_version","alias_value":"2110.14770v1","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.14770","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"pith_short_12","alias_value":"RHGU5DOB3MSE","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"pith_short_16","alias_value":"RHGU5DOB3MSEXSST","created_at":"2026-07-05T03:26:46Z"},{"alias_kind":"pith_short_8","alias_value":"RHGU5DOB","created_at":"2026-07-05T03:26:46Z"}],"graph_snapshots":[{"event_id":"sha256:9ae7c6bbc0e4a16fbc972f3ecd024aeda0435f9fecad46f6ab0e5e10548b4cf4","target":"graph","created_at":"2026-07-05T03:26:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2110.14770/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The aim in imitation learning is to learn effective policies by utilizing near-optimal expert demonstrations. However, high-quality demonstrations from human experts can be expensive to obtain in large numbers. On the other hand, it is often much easier to obtain large quantities of suboptimal or task-agnostic trajectories, which are not useful for direct imitation, but can nevertheless provide insight into the dynamical structure of the environment, showing what could be done in the environment even if not what should be done. We ask the question, is it possible to utilize such suboptimal off","authors_text":"Mengjiao Yang, Ofir Nachum, Sergey Levine","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-27T21:05:00Z","title":"TRAIL: Near-Optimal Imitation Learning with Suboptimal Data"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.14770","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:042a2a243c76722bc92e98cc050c13a58056e30e868363629a7704543b5a2339","target":"record","created_at":"2026-07-05T03:26:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"143f4fc7ece4b7e132094ec9392ea8852ffa7d2618f469a772a8865d9e92abe2","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-27T21:05:00Z","title_canon_sha256":"dadd884a872c1c205d08b8a7761cefae748d483315cc5652e8aaa7e86c66accf"},"schema_version":"1.0","source":{"id":"2110.14770","kind":"arxiv","version":1}},"canonical_sha256":"89cd4e8dc1db244bca53caf36f444cbefd4acee407708cce84c84b756061268f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"89cd4e8dc1db244bca53caf36f444cbefd4acee407708cce84c84b756061268f","first_computed_at":"2026-07-05T03:26:46.943864Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:26:46.943864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"aJECvN9pRV1460t/Whenj42w9TSOcylJttkVp1eymaUCe6yFKWMLA5qsv66eJcpEXRl9eq2q3g7B8Ju9qZjvAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T03:26:46.944320Z","signed_message":"canonical_sha256_bytes"},"source_id":"2110.14770","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:042a2a243c76722bc92e98cc050c13a58056e30e868363629a7704543b5a2339","sha256:9ae7c6bbc0e4a16fbc972f3ecd024aeda0435f9fecad46f6ab0e5e10548b4cf4"],"state_sha256":"474fcc6ad13e2855c1d83415f79519ac14da704167060c08957325782592435e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ubLG1yZRoHl52V2vUWqhPQQEz8xfnl7d9qDffBqoE+5k8qKhJkEkkkgelTsoQfx4hNd7CW9Gd4XB5ICaEB5WDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T18:51:08.253554Z","bundle_sha256":"d547b2663b73ac553d4a3a8afd386d6a33408c1c46f639c3b93cbd2fffaf1eee"}}