{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:JRYUPV57FEIEEY42VT6MVJUC7E","short_pith_number":"pith:JRYUPV57","canonical_record":{"source":{"id":"2310.19007","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-29T13:45:07Z","cross_cats_sorted":[],"title_canon_sha256":"1f73ab1369d79b8ae0a832bb021e2f1f1b58e7dc45337ff44956d7a45736cc90","abstract_canon_sha256":"1fbd2a1868a590c9e35d1c73007e013695821602ca033f121f0cb2d32e100caa"},"schema_version":"1.0"},"canonical_sha256":"4c7147d7bf291042639aacfccaa682f91279a31b9c27a438b3b2e1e9e73d3374","source":{"kind":"arxiv","id":"2310.19007","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.19007","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"arxiv_version","alias_value":"2310.19007v2","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19007","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"pith_short_12","alias_value":"JRYUPV57FEIE","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"pith_short_16","alias_value":"JRYUPV57FEIEEY42","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"pith_short_8","alias_value":"JRYUPV57","created_at":"2026-07-05T07:07:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:JRYUPV57FEIEEY42VT6MVJUC7E","target":"record","payload":{"canonical_record":{"source":{"id":"2310.19007","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-29T13:45:07Z","cross_cats_sorted":[],"title_canon_sha256":"1f73ab1369d79b8ae0a832bb021e2f1f1b58e7dc45337ff44956d7a45736cc90","abstract_canon_sha256":"1fbd2a1868a590c9e35d1c73007e013695821602ca033f121f0cb2d32e100caa"},"schema_version":"1.0"},"canonical_sha256":"4c7147d7bf291042639aacfccaa682f91279a31b9c27a438b3b2e1e9e73d3374","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:07:18.982989Z","signature_b64":"oAjTUjjJkAiN03UesRxeiZlOL1i9aIQ9iK17YP3F05pz4twTYVY9i2F7Zo4A2DDQo+nsaaT/3skYPSVA6GMxDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c7147d7bf291042639aacfccaa682f91279a31b9c27a438b3b2e1e9e73d3374","last_reissued_at":"2026-07-05T07:07:18.981601Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:07:18.981601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.19007","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:07:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"MkrZbM1d6Jk3giabvEq8qtcp9T2HxSh+Q2Lzd4P6fWTlpCQhKQxKTeXYhqCrpvKh7ibtg5TshnffTVanYPeICQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T04:52:11.505574Z"},"content_sha256":"70b725dd7de176432a74454a7a00f967defad7812e29cea069994fe41246ea90","schema_version":"1.0","event_id":"sha256:70b725dd7de176432a74454a7a00f967defad7812e29cea069994fe41246ea90"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:JRYUPV57FEIEEY42VT6MVJUC7E","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Behavior Alignment via Reward Function Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bruno Castro da Silva, Dhawal Gupta, Philip S. Thomas, Scott M. Jordan, Yash Chandak","submitted_at":"2023-10-29T13:45:07Z","abstract_excerpt":"Designing reward functions for efficiently guiding reinforcement learning (RL) agents toward specific behaviors is a complex task. This is challenging since it requires the identification of reward structures that are not sparse and that avoid inadvertently inducing undesirable behaviors. Naively modifying the reward structure to offer denser and more frequent feedback can lead to unintended outcomes and promote behaviors that are not aligned with the designer's intended goal. Although potential-based reward shaping is often suggested as a remedy, we systematically investigate settings where d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19007","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19007/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:07:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lf5r97H4KrzJ55bDQ342jGfUl5gfjPaDRc9Z+3peLM0Hc5WpyB12VZVba4I4aSCpvQ4RLxSzsrX2Da66UpdvCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T04:52:11.506670Z"},"content_sha256":"fb98b51501a9c0a44a8d295fec76334feca56725590580aa8604f107b085f3bf","schema_version":"1.0","event_id":"sha256:fb98b51501a9c0a44a8d295fec76334feca56725590580aa8604f107b085f3bf"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JRYUPV57FEIEEY42VT6MVJUC7E/bundle.json","state_url":"https://pith.science/pith/JRYUPV57FEIEEY42VT6MVJUC7E/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JRYUPV57FEIEEY42VT6MVJUC7E/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T04:52:11Z","links":{"resolver":"https://pith.science/pith/JRYUPV57FEIEEY42VT6MVJUC7E","bundle":"https://pith.science/pith/JRYUPV57FEIEEY42VT6MVJUC7E/bundle.json","state":"https://pith.science/pith/JRYUPV57FEIEEY42VT6MVJUC7E/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JRYUPV57FEIEEY42VT6MVJUC7E/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:JRYUPV57FEIEEY42VT6MVJUC7E","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1fbd2a1868a590c9e35d1c73007e013695821602ca033f121f0cb2d32e100caa","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-29T13:45:07Z","title_canon_sha256":"1f73ab1369d79b8ae0a832bb021e2f1f1b58e7dc45337ff44956d7a45736cc90"},"schema_version":"1.0","source":{"id":"2310.19007","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.19007","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"arxiv_version","alias_value":"2310.19007v2","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19007","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"pith_short_12","alias_value":"JRYUPV57FEIE","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"pith_short_16","alias_value":"JRYUPV57FEIEEY42","created_at":"2026-07-05T07:07:18Z"},{"alias_kind":"pith_short_8","alias_value":"JRYUPV57","created_at":"2026-07-05T07:07:18Z"}],"graph_snapshots":[{"event_id":"sha256:fb98b51501a9c0a44a8d295fec76334feca56725590580aa8604f107b085f3bf","target":"graph","created_at":"2026-07-05T07:07:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.19007/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Designing reward functions for efficiently guiding reinforcement learning (RL) agents toward specific behaviors is a complex task. This is challenging since it requires the identification of reward structures that are not sparse and that avoid inadvertently inducing undesirable behaviors. Naively modifying the reward structure to offer denser and more frequent feedback can lead to unintended outcomes and promote behaviors that are not aligned with the designer's intended goal. Although potential-based reward shaping is often suggested as a remedy, we systematically investigate settings where d","authors_text":"Bruno Castro da Silva, Dhawal Gupta, Philip S. Thomas, Scott M. Jordan, Yash Chandak","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-29T13:45:07Z","title":"Behavior Alignment via Reward Function Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19007","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:70b725dd7de176432a74454a7a00f967defad7812e29cea069994fe41246ea90","target":"record","created_at":"2026-07-05T07:07:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1fbd2a1868a590c9e35d1c73007e013695821602ca033f121f0cb2d32e100caa","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-29T13:45:07Z","title_canon_sha256":"1f73ab1369d79b8ae0a832bb021e2f1f1b58e7dc45337ff44956d7a45736cc90"},"schema_version":"1.0","source":{"id":"2310.19007","kind":"arxiv","version":2}},"canonical_sha256":"4c7147d7bf291042639aacfccaa682f91279a31b9c27a438b3b2e1e9e73d3374","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4c7147d7bf291042639aacfccaa682f91279a31b9c27a438b3b2e1e9e73d3374","first_computed_at":"2026-07-05T07:07:18.981601Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:07:18.981601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oAjTUjjJkAiN03UesRxeiZlOL1i9aIQ9iK17YP3F05pz4twTYVY9i2F7Zo4A2DDQo+nsaaT/3skYPSVA6GMxDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:07:18.982989Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.19007","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:70b725dd7de176432a74454a7a00f967defad7812e29cea069994fe41246ea90","sha256:fb98b51501a9c0a44a8d295fec76334feca56725590580aa8604f107b085f3bf"],"state_sha256":"d73a675cefca66601a6aebdfe30842416bde7dde671de40e825eaa1a41a2a981"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XqNdfY/WmCt9LKbO1kc2g9yadN9drRvH6+vP1ttcqB7KsRe5Ebcqw8B5iWdDpkU0jGBWCVlmMEfFIyJPiqKVDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T04:52:11.513940Z","bundle_sha256":"7abe91611260c418534c3a4f92ffdb41d1b79249bedda8eb77c11426ffaa77fd"}}