{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:SYB4PCOM34ETTJMZCHCZZPV5KJ","short_pith_number":"pith:SYB4PCOM","canonical_record":{"source":{"id":"2201.10081","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-25T03:48:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f0c1a034589c84367adc91e8c4031ab5923988d1161b73cf28e40c2dd1f90f6c","abstract_canon_sha256":"24cee139ad8a789cbc42779e72a173f3a1ad4615dda89b03252f98e0faa80a61"},"schema_version":"1.0"},"canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","source":{"kind":"arxiv","id":"2201.10081","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2201.10081","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"arxiv_version","alias_value":"2201.10081v1","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.10081","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"pith_short_12","alias_value":"SYB4PCOM34ET","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"pith_short_16","alias_value":"SYB4PCOM34ETTJMZ","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"pith_short_8","alias_value":"SYB4PCOM","created_at":"2026-07-05T03:51:15Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:SYB4PCOM34ETTJMZCHCZZPV5KJ","target":"record","payload":{"canonical_record":{"source":{"id":"2201.10081","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-25T03:48:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f0c1a034589c84367adc91e8c4031ab5923988d1161b73cf28e40c2dd1f90f6c","abstract_canon_sha256":"24cee139ad8a789cbc42779e72a173f3a1ad4615dda89b03252f98e0faa80a61"},"schema_version":"1.0"},"canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:51:15.307498Z","signature_b64":"ClfK9Ou5V/9ozaCYIuC89Gqzm72f+ikBgvw7p54ab5xj1zPHmCB057acggyQ3eA+Yv6DK9FwkJe8j0ppQdSHBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","last_reissued_at":"2026-07-05T03:51:15.306875Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:51:15.306875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2201.10081","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:51:15Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eBeMqFpIKWKBvOgaGDH1Q8dFQKT7qRcjBe14FpD27R/4nldS13a4YoKdPePgRvwFdeo6qXcz2fq1bG6JNZ9XAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T05:55:44.115282Z"},"content_sha256":"60e4bf2a6092c017fef0019537426a5fd02b6a35de2db45179327cbd2ca1726d","schema_version":"1.0","event_id":"sha256:60e4bf2a6092c017fef0019537426a5fd02b6a35de2db45179327cbd2ca1726d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:SYB4PCOM34ETTJMZCHCZZPV5KJ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Dynamics-Aware Comparison of Learned Reward Functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adrien Gaidon, Ashwin Balakrishna, Blake Wulfe, Jean Mercat, Logan Ellis, Rowan McAllister","submitted_at":"2022-01-25T03:48:00Z","abstract_excerpt":"The ability to learn reward functions plays an important role in enabling the deployment of intelligent agents in the real world. However, comparing reward functions, for example as a means of evaluating reward learning methods, presents a challenge. Reward functions are typically compared by considering the behavior of optimized policies, but this approach conflates deficiencies in the reward function with those of the policy search algorithm used to optimize it. To address this challenge, Gleave et al. (2020) propose the Equivalent-Policy Invariant Comparison (EPIC) distance. EPIC avoids pol"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.10081","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.10081/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:51:15Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eDz8bgv5wsbW71qUVV6MT88HWYw7An7Y3XIWz+7G+eq1Z0hqbFPo8y22bavpsXoI7UPFMgbFd/XNSvHNCqflBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T05:55:44.115776Z"},"content_sha256":"6ebcf88430396c97b325c1cc323919b61fbfcf06e5eeba76763e444884b6e05d","schema_version":"1.0","event_id":"sha256:6ebcf88430396c97b325c1cc323919b61fbfcf06e5eeba76763e444884b6e05d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/bundle.json","state_url":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T05:55:44Z","links":{"resolver":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ","bundle":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/bundle.json","state":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:SYB4PCOM34ETTJMZCHCZZPV5KJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"24cee139ad8a789cbc42779e72a173f3a1ad4615dda89b03252f98e0faa80a61","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-25T03:48:00Z","title_canon_sha256":"f0c1a034589c84367adc91e8c4031ab5923988d1161b73cf28e40c2dd1f90f6c"},"schema_version":"1.0","source":{"id":"2201.10081","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2201.10081","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"arxiv_version","alias_value":"2201.10081v1","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.10081","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"pith_short_12","alias_value":"SYB4PCOM34ET","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"pith_short_16","alias_value":"SYB4PCOM34ETTJMZ","created_at":"2026-07-05T03:51:15Z"},{"alias_kind":"pith_short_8","alias_value":"SYB4PCOM","created_at":"2026-07-05T03:51:15Z"}],"graph_snapshots":[{"event_id":"sha256:6ebcf88430396c97b325c1cc323919b61fbfcf06e5eeba76763e444884b6e05d","target":"graph","created_at":"2026-07-05T03:51:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2201.10081/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The ability to learn reward functions plays an important role in enabling the deployment of intelligent agents in the real world. However, comparing reward functions, for example as a means of evaluating reward learning methods, presents a challenge. Reward functions are typically compared by considering the behavior of optimized policies, but this approach conflates deficiencies in the reward function with those of the policy search algorithm used to optimize it. To address this challenge, Gleave et al. (2020) propose the Equivalent-Policy Invariant Comparison (EPIC) distance. EPIC avoids pol","authors_text":"Adrien Gaidon, Ashwin Balakrishna, Blake Wulfe, Jean Mercat, Logan Ellis, Rowan McAllister","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-25T03:48:00Z","title":"Dynamics-Aware Comparison of Learned Reward Functions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.10081","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:60e4bf2a6092c017fef0019537426a5fd02b6a35de2db45179327cbd2ca1726d","target":"record","created_at":"2026-07-05T03:51:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"24cee139ad8a789cbc42779e72a173f3a1ad4615dda89b03252f98e0faa80a61","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-25T03:48:00Z","title_canon_sha256":"f0c1a034589c84367adc91e8c4031ab5923988d1161b73cf28e40c2dd1f90f6c"},"schema_version":"1.0","source":{"id":"2201.10081","kind":"arxiv","version":1}},"canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","first_computed_at":"2026-07-05T03:51:15.306875Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:51:15.306875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ClfK9Ou5V/9ozaCYIuC89Gqzm72f+ikBgvw7p54ab5xj1zPHmCB057acggyQ3eA+Yv6DK9FwkJe8j0ppQdSHBg==","signature_status":"signed_v1","signed_at":"2026-07-05T03:51:15.307498Z","signed_message":"canonical_sha256_bytes"},"source_id":"2201.10081","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:60e4bf2a6092c017fef0019537426a5fd02b6a35de2db45179327cbd2ca1726d","sha256:6ebcf88430396c97b325c1cc323919b61fbfcf06e5eeba76763e444884b6e05d"],"state_sha256":"2611ca03ab17433f7d1c14246b18c11fce28c9af700e9e136e57e0eabd31c790"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6xnXdHBe0A5cnnNDksyy2b6j3fhEyXm3IGQyyOIfYvRPuzLwntW0Gh2Hj5Vf6OWDsj5r/6eHXeQXQf0aBIkKCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T05:55:44.119787Z","bundle_sha256":"5c2decc7157e4a979a56ecf9ba6c47f56bc324b0571187d7c98a1ee039ffe6bf"}}