{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:G32EXR7FN37QNHI7YFHUM75JL5","short_pith_number":"pith:G32EXR7F","canonical_record":{"source":{"id":"2405.15194","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-24T03:53:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cf253ae106aba377fd3bfea13334bc03117b2f60ab3cf65268ec5a5bc29d9ddf","abstract_canon_sha256":"c67764b78c984e0ccefc0d12a527fa8b104bf3a09321da942ff40a15f57bd50f"},"schema_version":"1.0"},"canonical_sha256":"36f44bc7e56eff069d1fc14f467fa95f41f28a263e2dc7eabac61ce2bd75ec00","source":{"kind":"arxiv","id":"2405.15194","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.15194","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"arxiv_version","alias_value":"2405.15194v2","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15194","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"pith_short_12","alias_value":"G32EXR7FN37Q","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"pith_short_16","alias_value":"G32EXR7FN37QNHI7","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"pith_short_8","alias_value":"G32EXR7F","created_at":"2026-07-05T09:17:13Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:G32EXR7FN37QNHI7YFHUM75JL5","target":"record","payload":{"canonical_record":{"source":{"id":"2405.15194","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-24T03:53:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cf253ae106aba377fd3bfea13334bc03117b2f60ab3cf65268ec5a5bc29d9ddf","abstract_canon_sha256":"c67764b78c984e0ccefc0d12a527fa8b104bf3a09321da942ff40a15f57bd50f"},"schema_version":"1.0"},"canonical_sha256":"36f44bc7e56eff069d1fc14f467fa95f41f28a263e2dc7eabac61ce2bd75ec00","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:13.462003Z","signature_b64":"ho2lnavnyvvZUCUq27nTSJLWbQ6MWFAgNs7hCnck2DHm7JiqzYSz3+0otLBEFajW3pkq58tNbFXCyHmT0BpLDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36f44bc7e56eff069d1fc14f467fa95f41f28a263e2dc7eabac61ce2bd75ec00","last_reissued_at":"2026-07-05T09:17:13.461440Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:13.461440Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.15194","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:17:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CIQRSKOSSAe1yd+/dydNjQc9hYJaLnUzfd9e26X8Rk8lZRoZJQWdWBWK6ViW2RQhrR20n2qDiny5PFu8jTIJAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:51:35.728931Z"},"content_sha256":"b198916ffc8d601699d235b89db25261806243687096bb9f9053cf90377619c1","schema_version":"1.0","event_id":"sha256:b198916ffc8d601699d235b89db25261806243687096bb9f9053cf90377619c1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:G32EXR7FN37QNHI7YFHUM75JL5","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amrita Bhattacharjee, Durgesh Kalwar, Huan Liu, Lin Guan, Siddhant Bhambri, Subbarao Kambhampati","submitted_at":"2024-05-24T03:53:57Z","abstract_excerpt":"Reinforcement Learning (RL) suffers from sample inefficiency in sparse reward domains, and the problem is further pronounced in case of stochastic transitions. To improve the sample efficiency, reward shaping is a well-studied approach to introduce intrinsic rewards that can help the RL agent converge to an optimal policy faster. However, designing a useful reward shaping function for all desirable states in the Markov Decision Process (MDP) is challenging, even for domain experts. Given that Large Language Models (LLMs) have demonstrated impressive performance across a magnitude of natural la"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15194","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:17:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xUeAoshpAB72gWBkr099hVgYGYoqR995MRrIuxKlfPFsoqTMGzJ8ZSO9Ox/lcxnadEuHKcb6x6qwYxcLKNMpAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:51:35.729564Z"},"content_sha256":"75a3657d95dd9e6bfc9d44414dd7de6666765fd026574daa00f6f334ed067772","schema_version":"1.0","event_id":"sha256:75a3657d95dd9e6bfc9d44414dd7de6666765fd026574daa00f6f334ed067772"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/G32EXR7FN37QNHI7YFHUM75JL5/bundle.json","state_url":"https://pith.science/pith/G32EXR7FN37QNHI7YFHUM75JL5/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/G32EXR7FN37QNHI7YFHUM75JL5/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T19:51:35Z","links":{"resolver":"https://pith.science/pith/G32EXR7FN37QNHI7YFHUM75JL5","bundle":"https://pith.science/pith/G32EXR7FN37QNHI7YFHUM75JL5/bundle.json","state":"https://pith.science/pith/G32EXR7FN37QNHI7YFHUM75JL5/state.json","well_known_bundle":"https://pith.science/.well-known/pith/G32EXR7FN37QNHI7YFHUM75JL5/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:G32EXR7FN37QNHI7YFHUM75JL5","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c67764b78c984e0ccefc0d12a527fa8b104bf3a09321da942ff40a15f57bd50f","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-24T03:53:57Z","title_canon_sha256":"cf253ae106aba377fd3bfea13334bc03117b2f60ab3cf65268ec5a5bc29d9ddf"},"schema_version":"1.0","source":{"id":"2405.15194","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.15194","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"arxiv_version","alias_value":"2405.15194v2","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15194","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"pith_short_12","alias_value":"G32EXR7FN37Q","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"pith_short_16","alias_value":"G32EXR7FN37QNHI7","created_at":"2026-07-05T09:17:13Z"},{"alias_kind":"pith_short_8","alias_value":"G32EXR7F","created_at":"2026-07-05T09:17:13Z"}],"graph_snapshots":[{"event_id":"sha256:75a3657d95dd9e6bfc9d44414dd7de6666765fd026574daa00f6f334ed067772","target":"graph","created_at":"2026-07-05T09:17:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.15194/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) suffers from sample inefficiency in sparse reward domains, and the problem is further pronounced in case of stochastic transitions. To improve the sample efficiency, reward shaping is a well-studied approach to introduce intrinsic rewards that can help the RL agent converge to an optimal policy faster. However, designing a useful reward shaping function for all desirable states in the Markov Decision Process (MDP) is challenging, even for domain experts. Given that Large Language Models (LLMs) have demonstrated impressive performance across a magnitude of natural la","authors_text":"Amrita Bhattacharjee, Durgesh Kalwar, Huan Liu, Lin Guan, Siddhant Bhambri, Subbarao Kambhampati","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-24T03:53:57Z","title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15194","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b198916ffc8d601699d235b89db25261806243687096bb9f9053cf90377619c1","target":"record","created_at":"2026-07-05T09:17:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c67764b78c984e0ccefc0d12a527fa8b104bf3a09321da942ff40a15f57bd50f","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-24T03:53:57Z","title_canon_sha256":"cf253ae106aba377fd3bfea13334bc03117b2f60ab3cf65268ec5a5bc29d9ddf"},"schema_version":"1.0","source":{"id":"2405.15194","kind":"arxiv","version":2}},"canonical_sha256":"36f44bc7e56eff069d1fc14f467fa95f41f28a263e2dc7eabac61ce2bd75ec00","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"36f44bc7e56eff069d1fc14f467fa95f41f28a263e2dc7eabac61ce2bd75ec00","first_computed_at":"2026-07-05T09:17:13.461440Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:17:13.461440Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ho2lnavnyvvZUCUq27nTSJLWbQ6MWFAgNs7hCnck2DHm7JiqzYSz3+0otLBEFajW3pkq58tNbFXCyHmT0BpLDw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:17:13.462003Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.15194","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b198916ffc8d601699d235b89db25261806243687096bb9f9053cf90377619c1","sha256:75a3657d95dd9e6bfc9d44414dd7de6666765fd026574daa00f6f334ed067772"],"state_sha256":"ad93c3a929bbdce621c58ab7843845885462c3ec2e18fb7210b0572b40d1ecc5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6v1dI7hIdD02ctGhXgP80Bfu851/utpnd3CCqHa2gP/wwPFnD+mPce4cw0cFEyvIR0Wb2nMFTtk5FOmh/ef4Dw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T19:51:35.739327Z","bundle_sha256":"19811b649eab045dfb18e47238e8bb5daeaf6322ba7bf1f3ec41d2a23ce1ca47"}}