{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SYB4PCOM34ETTJMZCHCZZPV5KJ","short_pith_number":"pith:SYB4PCOM","schema_version":"1.0","canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","source":{"kind":"arxiv","id":"2201.10081","version":1},"attestation_state":"computed","paper":{"title":"Dynamics-Aware Comparison of Learned Reward Functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adrien Gaidon, Ashwin Balakrishna, Blake Wulfe, Jean Mercat, Logan Ellis, Rowan McAllister","submitted_at":"2022-01-25T03:48:00Z","abstract_excerpt":"The ability to learn reward functions plays an important role in enabling the deployment of intelligent agents in the real world. However, comparing reward functions, for example as a means of evaluating reward learning methods, presents a challenge. Reward functions are typically compared by considering the behavior of optimized policies, but this approach conflates deficiencies in the reward function with those of the policy search algorithm used to optimize it. To address this challenge, Gleave et al. (2020) propose the Equivalent-Policy Invariant Comparison (EPIC) distance. EPIC avoids pol"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.10081","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-25T03:48:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f0c1a034589c84367adc91e8c4031ab5923988d1161b73cf28e40c2dd1f90f6c","abstract_canon_sha256":"24cee139ad8a789cbc42779e72a173f3a1ad4615dda89b03252f98e0faa80a61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:51:15.307498Z","signature_b64":"ClfK9Ou5V/9ozaCYIuC89Gqzm72f+ikBgvw7p54ab5xj1zPHmCB057acggyQ3eA+Yv6DK9FwkJe8j0ppQdSHBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9603c789ccdf0939a59911c59cbebd5245779b28d360f15e7889b4a1c151c6ab","last_reissued_at":"2026-07-05T03:51:15.306875Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:51:15.306875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamics-Aware Comparison of Learned Reward Functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adrien Gaidon, Ashwin Balakrishna, Blake Wulfe, Jean Mercat, Logan Ellis, Rowan McAllister","submitted_at":"2022-01-25T03:48:00Z","abstract_excerpt":"The ability to learn reward functions plays an important role in enabling the deployment of intelligent agents in the real world. However, comparing reward functions, for example as a means of evaluating reward learning methods, presents a challenge. Reward functions are typically compared by considering the behavior of optimized policies, but this approach conflates deficiencies in the reward function with those of the policy search algorithm used to optimize it. To address this challenge, Gleave et al. (2020) propose the Equivalent-Policy Invariant Comparison (EPIC) distance. EPIC avoids pol"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.10081","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.10081/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.10081","created_at":"2026-07-05T03:51:15.306938+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.10081v1","created_at":"2026-07-05T03:51:15.306938+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.10081","created_at":"2026-07-05T03:51:15.306938+00:00"},{"alias_kind":"pith_short_12","alias_value":"SYB4PCOM34ET","created_at":"2026-07-05T03:51:15.306938+00:00"},{"alias_kind":"pith_short_16","alias_value":"SYB4PCOM34ETTJMZ","created_at":"2026-07-05T03:51:15.306938+00:00"},{"alias_kind":"pith_short_8","alias_value":"SYB4PCOM","created_at":"2026-07-05T03:51:15.306938+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.15421","citing_title":"Reward Models in Deep Reinforcement Learning: A Survey","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ","json":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ.json","graph_json":"https://pith.science/api/pith-number/SYB4PCOM34ETTJMZCHCZZPV5KJ/graph.json","events_json":"https://pith.science/api/pith-number/SYB4PCOM34ETTJMZCHCZZPV5KJ/events.json","paper":"https://pith.science/paper/SYB4PCOM"},"agent_actions":{"view_html":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ","download_json":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ.json","view_paper":"https://pith.science/paper/SYB4PCOM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.10081&json=true","fetch_graph":"https://pith.science/api/pith-number/SYB4PCOM34ETTJMZCHCZZPV5KJ/graph.json","fetch_events":"https://pith.science/api/pith-number/SYB4PCOM34ETTJMZCHCZZPV5KJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/action/storage_attestation","attest_author":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/action/author_attestation","sign_citation":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/action/citation_signature","submit_replication":"https://pith.science/pith/SYB4PCOM34ETTJMZCHCZZPV5KJ/action/replication_record"}},"created_at":"2026-07-05T03:51:15.306938+00:00","updated_at":"2026-07-05T03:51:15.306938+00:00"}