{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:FVHDBY343HCDUNWKXYC6M6CU5H","short_pith_number":"pith:FVHDBY34","canonical_record":{"source":{"id":"2501.00381","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"7084df0c905380e686385a49984cc3c74204f85be3b7d8edf878ab323758e96d","abstract_canon_sha256":"ad93686f0f9cac27cee98b556b319c5c9bf18860be2a6e49c661257981c9b7c1"},"schema_version":"1.0"},"canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","source":{"kind":"arxiv","id":"2501.00381","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.00381","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"arxiv_version","alias_value":"2501.00381v1","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00381","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_12","alias_value":"FVHDBY343HCD","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_16","alias_value":"FVHDBY343HCDUNWK","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_8","alias_value":"FVHDBY34","created_at":"2026-07-05T09:55:44Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:FVHDBY343HCDUNWKXYC6M6CU5H","target":"record","payload":{"canonical_record":{"source":{"id":"2501.00381","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"7084df0c905380e686385a49984cc3c74204f85be3b7d8edf878ab323758e96d","abstract_canon_sha256":"ad93686f0f9cac27cee98b556b319c5c9bf18860be2a6e49c661257981c9b7c1"},"schema_version":"1.0"},"canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:44.223397Z","signature_b64":"dyvWhmTBBkCogQwjBQxMXHBYH0Ft7I1gD0j6yZ5Nc2f4NawBfbhwNidU4d2Md8cQQFEIGKpqTyj88Eh5szHUDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","last_reissued_at":"2026-07-05T09:55:44.222909Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:44.222909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2501.00381","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:55:44Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kW+OAtbAEbnJ7C+zopRzzvPtW1jPQQJ9Wu3949khhHH1gB25KJebZW8To9xyTXPHxiWnKDEyf30BpHVZK23OBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T03:22:33.257643Z"},"content_sha256":"11f196ab3ad59ad4501301fa124c3fa701bd508fbf3c9b1274c115f0339eac02","schema_version":"1.0","event_id":"sha256:11f196ab3ad59ad4501301fa124c3fa701bd508fbf3c9b1274c115f0339eac02"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:FVHDBY343HCDUNWKXYC6M6CU5H","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Toward Information Theoretic Active Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jack Golden, Jonathon Liu, Oliver Newcombe, Ondrej Bajgar, Rohan Narayan Langford Mitta, Sid William Gould","submitted_at":"2024-12-31T10:32:24Z","abstract_excerpt":"As AI systems become increasingly autonomous, aligning their decision-making to human preferences is essential. In domains like autonomous driving or robotics, it is impossible to write down the reward function representing these preferences by hand. Inverse reinforcement learning (IRL) offers a promising approach to infer the unknown reward from demonstrations. However, obtaining human demonstrations can be costly. Active IRL addresses this challenge by strategically selecting the most informative scenarios for human demonstration, reducing the amount of required human effort. Where most prio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00381","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.00381/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:55:44Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"g8ps25hV4lrzd9jReOgGopKrvG5K8C9TnKGvw7N3lX/ufQ9emkOHmpHGR5781Q8i8snWvGgdpejmkbSgBsfLAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T03:22:33.258292Z"},"content_sha256":"a238b7e1af39f6b669a1c65d2d23da8e2a8fb6cc08bcdd0f600d8323dec0879a","schema_version":"1.0","event_id":"sha256:a238b7e1af39f6b669a1c65d2d23da8e2a8fb6cc08bcdd0f600d8323dec0879a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FVHDBY343HCDUNWKXYC6M6CU5H/bundle.json","state_url":"https://pith.science/pith/FVHDBY343HCDUNWKXYC6M6CU5H/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FVHDBY343HCDUNWKXYC6M6CU5H/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-14T03:22:33Z","links":{"resolver":"https://pith.science/pith/FVHDBY343HCDUNWKXYC6M6CU5H","bundle":"https://pith.science/pith/FVHDBY343HCDUNWKXYC6M6CU5H/bundle.json","state":"https://pith.science/pith/FVHDBY343HCDUNWKXYC6M6CU5H/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FVHDBY343HCDUNWKXYC6M6CU5H/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:FVHDBY343HCDUNWKXYC6M6CU5H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ad93686f0f9cac27cee98b556b319c5c9bf18860be2a6e49c661257981c9b7c1","cross_cats_sorted":["stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","title_canon_sha256":"7084df0c905380e686385a49984cc3c74204f85be3b7d8edf878ab323758e96d"},"schema_version":"1.0","source":{"id":"2501.00381","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.00381","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"arxiv_version","alias_value":"2501.00381v1","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00381","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_12","alias_value":"FVHDBY343HCD","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_16","alias_value":"FVHDBY343HCDUNWK","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_8","alias_value":"FVHDBY34","created_at":"2026-07-05T09:55:44Z"}],"graph_snapshots":[{"event_id":"sha256:a238b7e1af39f6b669a1c65d2d23da8e2a8fb6cc08bcdd0f600d8323dec0879a","target":"graph","created_at":"2026-07-05T09:55:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2501.00381/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As AI systems become increasingly autonomous, aligning their decision-making to human preferences is essential. In domains like autonomous driving or robotics, it is impossible to write down the reward function representing these preferences by hand. Inverse reinforcement learning (IRL) offers a promising approach to infer the unknown reward from demonstrations. However, obtaining human demonstrations can be costly. Active IRL addresses this challenge by strategically selecting the most informative scenarios for human demonstration, reducing the amount of required human effort. Where most prio","authors_text":"Jack Golden, Jonathon Liu, Oliver Newcombe, Ondrej Bajgar, Rohan Narayan Langford Mitta, Sid William Gould","cross_cats":["stat.ML"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","title":"Toward Information Theoretic Active Inverse Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00381","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:11f196ab3ad59ad4501301fa124c3fa701bd508fbf3c9b1274c115f0339eac02","target":"record","created_at":"2026-07-05T09:55:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ad93686f0f9cac27cee98b556b319c5c9bf18860be2a6e49c661257981c9b7c1","cross_cats_sorted":["stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","title_canon_sha256":"7084df0c905380e686385a49984cc3c74204f85be3b7d8edf878ab323758e96d"},"schema_version":"1.0","source":{"id":"2501.00381","kind":"arxiv","version":1}},"canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","first_computed_at":"2026-07-05T09:55:44.222909Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:55:44.222909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"dyvWhmTBBkCogQwjBQxMXHBYH0Ft7I1gD0j6yZ5Nc2f4NawBfbhwNidU4d2Md8cQQFEIGKpqTyj88Eh5szHUDA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:55:44.223397Z","signed_message":"canonical_sha256_bytes"},"source_id":"2501.00381","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:11f196ab3ad59ad4501301fa124c3fa701bd508fbf3c9b1274c115f0339eac02","sha256:a238b7e1af39f6b669a1c65d2d23da8e2a8fb6cc08bcdd0f600d8323dec0879a"],"state_sha256":"8ee7078ac92d219dc7bf572c0dd18bfa087acbd1ccdfd6bb51de0c2b88134410"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FKU7j2NaOlNOV/u5nXP6sANYZxlCKRVUCCJKts5N61BFsaR7A9pD6HMbjRtHGl8zGQoQpohx3PSTN/BCaaUfBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-14T03:22:33.263960Z","bundle_sha256":"53aa426c00a9d26438c9907bd23dd21337dd7c51bc35fef3c1386b3bea6c0b5d"}}