{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:LCTU3VPEUXGC5NXMA2Q6EF3ZML","short_pith_number":"pith:LCTU3VPE","canonical_record":{"source":{"id":"2102.06483","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-12T12:32:02Z","cross_cats_sorted":[],"title_canon_sha256":"b8e7110c9fa35608cfe143c0fbf436c5cbb14bf6b886c3e2fe8525de4d51c427","abstract_canon_sha256":"0391d97a57b46bbd3a46da7e486324a935627bfefc83f47c3df82416b739cbc7"},"schema_version":"1.0"},"canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","source":{"kind":"arxiv","id":"2102.06483","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2102.06483","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"arxiv_version","alias_value":"2102.06483v2","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.06483","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"pith_short_12","alias_value":"LCTU3VPEUXGC","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"pith_short_16","alias_value":"LCTU3VPEUXGC5NXM","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"pith_short_8","alias_value":"LCTU3VPE","created_at":"2026-07-05T02:22:25Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:LCTU3VPEUXGC5NXMA2Q6EF3ZML","target":"record","payload":{"canonical_record":{"source":{"id":"2102.06483","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-12T12:32:02Z","cross_cats_sorted":[],"title_canon_sha256":"b8e7110c9fa35608cfe143c0fbf436c5cbb14bf6b886c3e2fe8525de4d51c427","abstract_canon_sha256":"0391d97a57b46bbd3a46da7e486324a935627bfefc83f47c3df82416b739cbc7"},"schema_version":"1.0"},"canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:22:25.705919Z","signature_b64":"1dMzr2VeOQF44U4w5akswpD+igowvqE0bmG1uxkTYQNfV+PsXZg2M3ilwWHSyft2ivdUQR17FT2emC/FgcgoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","last_reissued_at":"2026-07-05T02:22:25.705513Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:22:25.705513Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2102.06483","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:22:25Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xjZ7OcA2O5r64tZ5/gOSfn2YA3DTSAZzpx5f6xMb5wrS66PpGgEpuC+mb3O2J6li4PTNqgvC3A69Jgx6R1jqBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T01:05:17.726217Z"},"content_sha256":"98e706d4b1a18fcc639de58b60b99befb2e4cfa799bd790ac7aa7d9c7d5322f1","schema_version":"1.0","event_id":"sha256:98e706d4b1a18fcc639de58b60b99befb2e4cfa799bd790ac7aa7d9c7d5322f1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:LCTU3VPEUXGC5NXMA2Q6EF3ZML","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Scalable Bayesian Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Mihaela van der Schaar","submitted_at":"2021-02-12T12:32:02Z","abstract_excerpt":"Bayesian inference over the reward presents an ideal solution to the ill-posed nature of the inverse reinforcement learning problem. Unfortunately current methods generally do not scale well beyond the small tabular setting due to the need for an inner-loop MDP solver, and even non-Bayesian methods that do themselves scale often require extensive interaction with the environment to perform well, being inappropriate for high stakes or costly applications such as healthcare. In this paper we introduce our method, Approximate Variational Reward Imitation Learning (AVRIL), that addresses both of t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.06483","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.06483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:22:25Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lcTnEo9bXib4SlVciA4yBLlufpWn4Dy4bjzOmp3QKErEkhUB+ns8xgk8G0f5vcFuxKEo8014KZ0o2A8V39CtBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T01:05:17.727327Z"},"content_sha256":"f397d667a46a5de675b3bbda7d55601859dd1bca116cd2ed81f804fa49638866","schema_version":"1.0","event_id":"sha256:f397d667a46a5de675b3bbda7d55601859dd1bca116cd2ed81f804fa49638866"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/bundle.json","state_url":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T01:05:17Z","links":{"resolver":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML","bundle":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/bundle.json","state":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/state.json","well_known_bundle":"https://pith.science/.well-known/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:LCTU3VPEUXGC5NXMA2Q6EF3ZML","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0391d97a57b46bbd3a46da7e486324a935627bfefc83f47c3df82416b739cbc7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-12T12:32:02Z","title_canon_sha256":"b8e7110c9fa35608cfe143c0fbf436c5cbb14bf6b886c3e2fe8525de4d51c427"},"schema_version":"1.0","source":{"id":"2102.06483","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2102.06483","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"arxiv_version","alias_value":"2102.06483v2","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.06483","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"pith_short_12","alias_value":"LCTU3VPEUXGC","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"pith_short_16","alias_value":"LCTU3VPEUXGC5NXM","created_at":"2026-07-05T02:22:25Z"},{"alias_kind":"pith_short_8","alias_value":"LCTU3VPE","created_at":"2026-07-05T02:22:25Z"}],"graph_snapshots":[{"event_id":"sha256:f397d667a46a5de675b3bbda7d55601859dd1bca116cd2ed81f804fa49638866","target":"graph","created_at":"2026-07-05T02:22:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2102.06483/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Bayesian inference over the reward presents an ideal solution to the ill-posed nature of the inverse reinforcement learning problem. Unfortunately current methods generally do not scale well beyond the small tabular setting due to the need for an inner-loop MDP solver, and even non-Bayesian methods that do themselves scale often require extensive interaction with the environment to perform well, being inappropriate for high stakes or costly applications such as healthcare. In this paper we introduce our method, Approximate Variational Reward Imitation Learning (AVRIL), that addresses both of t","authors_text":"Alex J. Chan, Mihaela van der Schaar","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-12T12:32:02Z","title":"Scalable Bayesian Inverse Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.06483","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:98e706d4b1a18fcc639de58b60b99befb2e4cfa799bd790ac7aa7d9c7d5322f1","target":"record","created_at":"2026-07-05T02:22:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0391d97a57b46bbd3a46da7e486324a935627bfefc83f47c3df82416b739cbc7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-12T12:32:02Z","title_canon_sha256":"b8e7110c9fa35608cfe143c0fbf436c5cbb14bf6b886c3e2fe8525de4d51c427"},"schema_version":"1.0","source":{"id":"2102.06483","kind":"arxiv","version":2}},"canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","first_computed_at":"2026-07-05T02:22:25.705513Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:22:25.705513Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"1dMzr2VeOQF44U4w5akswpD+igowvqE0bmG1uxkTYQNfV+PsXZg2M3ilwWHSyft2ivdUQR17FT2emC/FgcgoCg==","signature_status":"signed_v1","signed_at":"2026-07-05T02:22:25.705919Z","signed_message":"canonical_sha256_bytes"},"source_id":"2102.06483","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:98e706d4b1a18fcc639de58b60b99befb2e4cfa799bd790ac7aa7d9c7d5322f1","sha256:f397d667a46a5de675b3bbda7d55601859dd1bca116cd2ed81f804fa49638866"],"state_sha256":"d56dec9b237203d7988ee102ddaf255c3c8188d8dbde8b7ff724ee1f25f047e0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Vmc+SCQ7Mh9IdYfJH+tC9XVB2ZtQcpet5ekfomAKeAluIgxSoVMMChJCtXnrwZ8+WYGC59MSdG9YB5fiZf+cAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T01:05:17.734118Z","bundle_sha256":"23550fd3ceca6fa0a1670cd1165866ce4395fb5c97c47b685e67b0c1f854298d"}}