{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XTZR3TCQDQAKPBZDLV3JN5N2NN","short_pith_number":"pith:XTZR3TCQ","schema_version":"1.0","canonical_sha256":"bcf31dcc501c00a787235d7696f5ba6b466bac6c11c82486383c3c58d5ba4d62","source":{"kind":"arxiv","id":"2403.08955","version":4},"attestation_state":"computed","paper":{"title":"Towards Efficient Risk-Sensitive Policy Gradient: An Iteration Complexity Analysis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Anish Gupta, Erfaun Noorani, Pratap Tokekar, Rui Liu","submitted_at":"2024-03-13T20:50:49Z","abstract_excerpt":"Reinforcement Learning (RL) has shown exceptional performance across various applications, enabling autonomous agents to learn optimal policies through interaction with their environments. However, traditional RL frameworks often face challenges in terms of iteration efficiency and safety. Risk-sensitive policy gradient methods, which incorporate both expected return and risk measures, have been explored for their ability to yield safe policies, yet their iteration complexity remains largely underexplored. In this work, we conduct a rigorous iteration complexity analysis for the risk-sensitive"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08955","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-13T20:50:49Z","cross_cats_sorted":["cs.AI","math.OC"],"title_canon_sha256":"df478decd7f3aaf0af59717598c77a507a48d78763733ee78b5ea1b0088d7c26","abstract_canon_sha256":"a3f3873f7cd01af2b1adcab9495e447d81fea42d87a2d013fca6aa9d55bc280f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:53.444462Z","signature_b64":"tuy9qTE4EFzfaF/TzqClZBiTCe58hleemwTr0QgTbhIvWBWvv57Hdr04rMNRV+kd67kBehyD5paTZh5n7mkBCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bcf31dcc501c00a787235d7696f5ba6b466bac6c11c82486383c3c58d5ba4d62","last_reissued_at":"2026-07-05T12:01:53.443983Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:53.443983Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Efficient Risk-Sensitive Policy Gradient: An Iteration Complexity Analysis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Anish Gupta, Erfaun Noorani, Pratap Tokekar, Rui Liu","submitted_at":"2024-03-13T20:50:49Z","abstract_excerpt":"Reinforcement Learning (RL) has shown exceptional performance across various applications, enabling autonomous agents to learn optimal policies through interaction with their environments. However, traditional RL frameworks often face challenges in terms of iteration efficiency and safety. Risk-sensitive policy gradient methods, which incorporate both expected return and risk measures, have been explored for their ability to yield safe policies, yet their iteration complexity remains largely underexplored. In this work, we conduct a rigorous iteration complexity analysis for the risk-sensitive"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08955","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08955/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08955","created_at":"2026-07-05T12:01:53.444048+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08955v4","created_at":"2026-07-05T12:01:53.444048+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08955","created_at":"2026-07-05T12:01:53.444048+00:00"},{"alias_kind":"pith_short_12","alias_value":"XTZR3TCQDQAK","created_at":"2026-07-05T12:01:53.444048+00:00"},{"alias_kind":"pith_short_16","alias_value":"XTZR3TCQDQAKPBZD","created_at":"2026-07-05T12:01:53.444048+00:00"},{"alias_kind":"pith_short_8","alias_value":"XTZR3TCQ","created_at":"2026-07-05T12:01:53.444048+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN","json":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN.json","graph_json":"https://pith.science/api/pith-number/XTZR3TCQDQAKPBZDLV3JN5N2NN/graph.json","events_json":"https://pith.science/api/pith-number/XTZR3TCQDQAKPBZDLV3JN5N2NN/events.json","paper":"https://pith.science/paper/XTZR3TCQ"},"agent_actions":{"view_html":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN","download_json":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN.json","view_paper":"https://pith.science/paper/XTZR3TCQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08955&json=true","fetch_graph":"https://pith.science/api/pith-number/XTZR3TCQDQAKPBZDLV3JN5N2NN/graph.json","fetch_events":"https://pith.science/api/pith-number/XTZR3TCQDQAKPBZDLV3JN5N2NN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN/action/storage_attestation","attest_author":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN/action/author_attestation","sign_citation":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN/action/citation_signature","submit_replication":"https://pith.science/pith/XTZR3TCQDQAKPBZDLV3JN5N2NN/action/replication_record"}},"created_at":"2026-07-05T12:01:53.444048+00:00","updated_at":"2026-07-05T12:01:53.444048+00:00"}