{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:FVHDBY343HCDUNWKXYC6M6CU5H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ad93686f0f9cac27cee98b556b319c5c9bf18860be2a6e49c661257981c9b7c1","cross_cats_sorted":["stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","title_canon_sha256":"7084df0c905380e686385a49984cc3c74204f85be3b7d8edf878ab323758e96d"},"schema_version":"1.0","source":{"id":"2501.00381","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.00381","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"arxiv_version","alias_value":"2501.00381v1","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00381","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_12","alias_value":"FVHDBY343HCD","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_16","alias_value":"FVHDBY343HCDUNWK","created_at":"2026-07-05T09:55:44Z"},{"alias_kind":"pith_short_8","alias_value":"FVHDBY34","created_at":"2026-07-05T09:55:44Z"}],"graph_snapshots":[{"event_id":"sha256:a238b7e1af39f6b669a1c65d2d23da8e2a8fb6cc08bcdd0f600d8323dec0879a","target":"graph","created_at":"2026-07-05T09:55:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2501.00381/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As AI systems become increasingly autonomous, aligning their decision-making to human preferences is essential. In domains like autonomous driving or robotics, it is impossible to write down the reward function representing these preferences by hand. Inverse reinforcement learning (IRL) offers a promising approach to infer the unknown reward from demonstrations. However, obtaining human demonstrations can be costly. Active IRL addresses this challenge by strategically selecting the most informative scenarios for human demonstration, reducing the amount of required human effort. Where most prio","authors_text":"Jack Golden, Jonathon Liu, Oliver Newcombe, Ondrej Bajgar, Rohan Narayan Langford Mitta, Sid William Gould","cross_cats":["stat.ML"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","title":"Toward Information Theoretic Active Inverse Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00381","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:11f196ab3ad59ad4501301fa124c3fa701bd508fbf3c9b1274c115f0339eac02","target":"record","created_at":"2026-07-05T09:55:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ad93686f0f9cac27cee98b556b319c5c9bf18860be2a6e49c661257981c9b7c1","cross_cats_sorted":["stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T10:32:24Z","title_canon_sha256":"7084df0c905380e686385a49984cc3c74204f85be3b7d8edf878ab323758e96d"},"schema_version":"1.0","source":{"id":"2501.00381","kind":"arxiv","version":1}},"canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2d4e30e37cd9c43a36cabe05e67854e9f642a36fc0d48b35a629835e0af16187","first_computed_at":"2026-07-05T09:55:44.222909Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:55:44.222909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"dyvWhmTBBkCogQwjBQxMXHBYH0Ft7I1gD0j6yZ5Nc2f4NawBfbhwNidU4d2Md8cQQFEIGKpqTyj88Eh5szHUDA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:55:44.223397Z","signed_message":"canonical_sha256_bytes"},"source_id":"2501.00381","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:11f196ab3ad59ad4501301fa124c3fa701bd508fbf3c9b1274c115f0339eac02","sha256:a238b7e1af39f6b669a1c65d2d23da8e2a8fb6cc08bcdd0f600d8323dec0879a"],"state_sha256":"8ee7078ac92d219dc7bf572c0dd18bfa087acbd1ccdfd6bb51de0c2b88134410"}