{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:37JR44VDWVW4Z3XTMQJE4CBK6E","short_pith_number":"pith:37JR44VD","schema_version":"1.0","canonical_sha256":"dfd31e72a3b56dcceef364124e082af12e80144b950ca4d49a97f24a6b7cdf86","source":{"kind":"arxiv","id":"2307.01452","version":2},"attestation_state":"computed","paper":{"title":"Causal Reinforcement Learning: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chengqi Zhang, Guodong Long, Jing Jiang, Zhihong Deng","submitted_at":"2023-07-04T03:00:43Z","abstract_excerpt":"Reinforcement learning is an essential paradigm for solving sequential decision problems under uncertainty. Despite many remarkable achievements in recent decades, applying reinforcement learning methods in the real world remains challenging. One of the main obstacles is that reinforcement learning agents lack a fundamental understanding of the world and must therefore learn from scratch through numerous trial-and-error interactions. They may also face challenges in providing explanations for their decisions and generalizing the acquired knowledge. Causality, however, offers a notable advantag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.01452","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-04T03:00:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0648e6066805dbae51f09ab051a7f07a9f4884fe12691b8eef126fa4f126c447","abstract_canon_sha256":"056f07be9de327ee21c3c798b90804f4801595bf70fd2641219ba6139ed399df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:54.855095Z","signature_b64":"AcUoj4sGr1sHFQy2wqqpXeNFBDopTU2V45kzfYqhl3VxXwWkYTUvDlnX3VxOBBPoQx99cdpWT+FB7M/m0C8aAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dfd31e72a3b56dcceef364124e082af12e80144b950ca4d49a97f24a6b7cdf86","last_reissued_at":"2026-07-05T07:14:54.854611Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:54.854611Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal Reinforcement Learning: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chengqi Zhang, Guodong Long, Jing Jiang, Zhihong Deng","submitted_at":"2023-07-04T03:00:43Z","abstract_excerpt":"Reinforcement learning is an essential paradigm for solving sequential decision problems under uncertainty. Despite many remarkable achievements in recent decades, applying reinforcement learning methods in the real world remains challenging. One of the main obstacles is that reinforcement learning agents lack a fundamental understanding of the world and must therefore learn from scratch through numerous trial-and-error interactions. They may also face challenges in providing explanations for their decisions and generalizing the acquired knowledge. Causality, however, offers a notable advantag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.01452","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.01452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.01452","created_at":"2026-07-05T07:14:54.854668+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.01452v2","created_at":"2026-07-05T07:14:54.854668+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.01452","created_at":"2026-07-05T07:14:54.854668+00:00"},{"alias_kind":"pith_short_12","alias_value":"37JR44VDWVW4","created_at":"2026-07-05T07:14:54.854668+00:00"},{"alias_kind":"pith_short_16","alias_value":"37JR44VDWVW4Z3XT","created_at":"2026-07-05T07:14:54.854668+00:00"},{"alias_kind":"pith_short_8","alias_value":"37JR44VD","created_at":"2026-07-05T07:14:54.854668+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16054","citing_title":"Ada-Diffuser: Latent-Aware Adaptive Diffusion for Decision-Making","ref_index":233,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E","json":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E.json","graph_json":"https://pith.science/api/pith-number/37JR44VDWVW4Z3XTMQJE4CBK6E/graph.json","events_json":"https://pith.science/api/pith-number/37JR44VDWVW4Z3XTMQJE4CBK6E/events.json","paper":"https://pith.science/paper/37JR44VD"},"agent_actions":{"view_html":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E","download_json":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E.json","view_paper":"https://pith.science/paper/37JR44VD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.01452&json=true","fetch_graph":"https://pith.science/api/pith-number/37JR44VDWVW4Z3XTMQJE4CBK6E/graph.json","fetch_events":"https://pith.science/api/pith-number/37JR44VDWVW4Z3XTMQJE4CBK6E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E/action/storage_attestation","attest_author":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E/action/author_attestation","sign_citation":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E/action/citation_signature","submit_replication":"https://pith.science/pith/37JR44VDWVW4Z3XTMQJE4CBK6E/action/replication_record"}},"created_at":"2026-07-05T07:14:54.854668+00:00","updated_at":"2026-07-05T07:14:54.854668+00:00"}