{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NY5J3X2RVZHD5I6JMBDM5MN3NX","short_pith_number":"pith:NY5J3X2R","schema_version":"1.0","canonical_sha256":"6e3a9ddf51ae4e3ea3c96046ceb1bb6df27d775a9ddb37d907fbf5d2bb2bd9f3","source":{"kind":"arxiv","id":"2312.01072","version":2},"attestation_state":"computed","paper":{"title":"A Survey of Temporal Credit Assignment in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Eduardo Pignatelli, Hado van Hasselt, Johan Ferret, Laura Toni, Matthieu Geist, Olivier Pietquin, Thomas Mesnard","submitted_at":"2023-12-02T08:49:51Z","abstract_excerpt":"The Credit Assignment Problem (CAP) refers to the longstanding challenge of Reinforcement Learning (RL) agents to associate actions with their long-term consequences. Solving the CAP is a crucial step towards the successful deployment of RL in the real world since most decision problems provide feedback that is noisy, delayed, and with little or no information about the causes. These conditions make it hard to distinguish serendipitous outcomes from those caused by informed decision-making. However, the mathematical nature of credit and the CAP remains poorly understood and defined. In this su"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.01072","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-02T08:49:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3d34a020c04419a76a42a1f3f1dc47c6a4712b7ef003e2bf6e51f98a70b6f996","abstract_canon_sha256":"bb95c5618e6c02c9df629fbea2dbb643eba40dafd066f893b4a51ce717810865"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:10.870921Z","signature_b64":"1tCCsNK0SPXLoe5FOOcYvj5gdwoffU0lKKSkUnY/vWEFZSKxB8XOvOQpoHd7XmwB8iUz2oEF5G29RG29Z7PiBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e3a9ddf51ae4e3ea3c96046ceb1bb6df27d775a9ddb37d907fbf5d2bb2bd9f3","last_reissued_at":"2026-07-05T08:40:10.870521Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:10.870521Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Temporal Credit Assignment in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Eduardo Pignatelli, Hado van Hasselt, Johan Ferret, Laura Toni, Matthieu Geist, Olivier Pietquin, Thomas Mesnard","submitted_at":"2023-12-02T08:49:51Z","abstract_excerpt":"The Credit Assignment Problem (CAP) refers to the longstanding challenge of Reinforcement Learning (RL) agents to associate actions with their long-term consequences. Solving the CAP is a crucial step towards the successful deployment of RL in the real world since most decision problems provide feedback that is noisy, delayed, and with little or no information about the causes. These conditions make it hard to distinguish serendipitous outcomes from those caused by informed decision-making. However, the mathematical nature of credit and the CAP remains poorly understood and defined. In this su"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.01072","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.01072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.01072","created_at":"2026-07-05T08:40:10.870577+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.01072v2","created_at":"2026-07-05T08:40:10.870577+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.01072","created_at":"2026-07-05T08:40:10.870577+00:00"},{"alias_kind":"pith_short_12","alias_value":"NY5J3X2RVZHD","created_at":"2026-07-05T08:40:10.870577+00:00"},{"alias_kind":"pith_short_16","alias_value":"NY5J3X2RVZHD5I6J","created_at":"2026-07-05T08:40:10.870577+00:00"},{"alias_kind":"pith_short_8","alias_value":"NY5J3X2R","created_at":"2026-07-05T08:40:10.870577+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24742","citing_title":"World Value Models for Robotic Manipulation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05885","citing_title":"When Denser Credit Is Not Enough: Evidence-Calibrated Policy Optimization for Long-Horizon LLM Agent Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04889","citing_title":"GRAIL: Gradient-Reweighted Advantages for Reinforcement Learning with Verifiable Rewards","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02355","citing_title":"SIRI: Self-Internalizing Reinforcement Learning with Intrinsic Skills for LLM Agent Training","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24005","citing_title":"LC-ERD: Mining Latent Logic for Self-Evolving Reasoning via Consistency-Regulated Reward Decomposition","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14558","citing_title":"Resolving Action Bottleneck: Agentic Reinforcement Learning Informed by Token-Level Energy","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03023","citing_title":"Behavior-Constrained Reinforcement Learning with Receding-Horizon Credit Assignment for High-Performance Control","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01954","citing_title":"Moira: Language-driven Hierarchical Reinforcement Learning for Pair Trading","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09459","citing_title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX","json":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX.json","graph_json":"https://pith.science/api/pith-number/NY5J3X2RVZHD5I6JMBDM5MN3NX/graph.json","events_json":"https://pith.science/api/pith-number/NY5J3X2RVZHD5I6JMBDM5MN3NX/events.json","paper":"https://pith.science/paper/NY5J3X2R"},"agent_actions":{"view_html":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX","download_json":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX.json","view_paper":"https://pith.science/paper/NY5J3X2R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.01072&json=true","fetch_graph":"https://pith.science/api/pith-number/NY5J3X2RVZHD5I6JMBDM5MN3NX/graph.json","fetch_events":"https://pith.science/api/pith-number/NY5J3X2RVZHD5I6JMBDM5MN3NX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX/action/storage_attestation","attest_author":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX/action/author_attestation","sign_citation":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX/action/citation_signature","submit_replication":"https://pith.science/pith/NY5J3X2RVZHD5I6JMBDM5MN3NX/action/replication_record"}},"created_at":"2026-07-05T08:40:10.870577+00:00","updated_at":"2026-07-05T08:40:10.870577+00:00"}