{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LCTU3VPEUXGC5NXMA2Q6EF3ZML","short_pith_number":"pith:LCTU3VPE","schema_version":"1.0","canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","source":{"kind":"arxiv","id":"2102.06483","version":2},"attestation_state":"computed","paper":{"title":"Scalable Bayesian Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Mihaela van der Schaar","submitted_at":"2021-02-12T12:32:02Z","abstract_excerpt":"Bayesian inference over the reward presents an ideal solution to the ill-posed nature of the inverse reinforcement learning problem. Unfortunately current methods generally do not scale well beyond the small tabular setting due to the need for an inner-loop MDP solver, and even non-Bayesian methods that do themselves scale often require extensive interaction with the environment to perform well, being inappropriate for high stakes or costly applications such as healthcare. In this paper we introduce our method, Approximate Variational Reward Imitation Learning (AVRIL), that addresses both of t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.06483","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-12T12:32:02Z","cross_cats_sorted":[],"title_canon_sha256":"b8e7110c9fa35608cfe143c0fbf436c5cbb14bf6b886c3e2fe8525de4d51c427","abstract_canon_sha256":"0391d97a57b46bbd3a46da7e486324a935627bfefc83f47c3df82416b739cbc7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:22:25.705919Z","signature_b64":"1dMzr2VeOQF44U4w5akswpD+igowvqE0bmG1uxkTYQNfV+PsXZg2M3ilwWHSyft2ivdUQR17FT2emC/FgcgoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58a74dd5e4a5cc2eb6ec06a1e2177962fff3b352f2d1c0e3f9924ece16106a1d","last_reissued_at":"2026-07-05T02:22:25.705513Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:22:25.705513Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Bayesian Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Mihaela van der Schaar","submitted_at":"2021-02-12T12:32:02Z","abstract_excerpt":"Bayesian inference over the reward presents an ideal solution to the ill-posed nature of the inverse reinforcement learning problem. Unfortunately current methods generally do not scale well beyond the small tabular setting due to the need for an inner-loop MDP solver, and even non-Bayesian methods that do themselves scale often require extensive interaction with the environment to perform well, being inappropriate for high stakes or costly applications such as healthcare. In this paper we introduce our method, Approximate Variational Reward Imitation Learning (AVRIL), that addresses both of t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.06483","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.06483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.06483","created_at":"2026-07-05T02:22:25.705571+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.06483v2","created_at":"2026-07-05T02:22:25.705571+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.06483","created_at":"2026-07-05T02:22:25.705571+00:00"},{"alias_kind":"pith_short_12","alias_value":"LCTU3VPEUXGC","created_at":"2026-07-05T02:22:25.705571+00:00"},{"alias_kind":"pith_short_16","alias_value":"LCTU3VPEUXGC5NXM","created_at":"2026-07-05T02:22:25.705571+00:00"},{"alias_kind":"pith_short_8","alias_value":"LCTU3VPE","created_at":"2026-07-05T02:22:25.705571+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.22966","citing_title":"Detection of coordinated fleet vehicles in route choice urban games. Part I. Inverse fleet assignment theory","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML","json":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML.json","graph_json":"https://pith.science/api/pith-number/LCTU3VPEUXGC5NXMA2Q6EF3ZML/graph.json","events_json":"https://pith.science/api/pith-number/LCTU3VPEUXGC5NXMA2Q6EF3ZML/events.json","paper":"https://pith.science/paper/LCTU3VPE"},"agent_actions":{"view_html":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML","download_json":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML.json","view_paper":"https://pith.science/paper/LCTU3VPE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.06483&json=true","fetch_graph":"https://pith.science/api/pith-number/LCTU3VPEUXGC5NXMA2Q6EF3ZML/graph.json","fetch_events":"https://pith.science/api/pith-number/LCTU3VPEUXGC5NXMA2Q6EF3ZML/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/action/storage_attestation","attest_author":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/action/author_attestation","sign_citation":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/action/citation_signature","submit_replication":"https://pith.science/pith/LCTU3VPEUXGC5NXMA2Q6EF3ZML/action/replication_record"}},"created_at":"2026-07-05T02:22:25.705571+00:00","updated_at":"2026-07-05T02:22:25.705571+00:00"}