{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LXHGWZODQPQ7PX3YW3TLQZHLL6","short_pith_number":"pith:LXHGWZOD","schema_version":"1.0","canonical_sha256":"5dce6b65c383e1f7df78b6e6b864eb5fb07c323a07fa7e6962746d05c8460f09","source":{"kind":"arxiv","id":"2408.07660","version":1},"attestation_state":"computed","paper":{"title":"Off-Policy Reinforcement Learning with High Dimensional Reward","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Dong Neuck Lee, Michael R. Kosorok","submitted_at":"2024-08-14T16:44:56Z","abstract_excerpt":"Conventional off-policy reinforcement learning (RL) focuses on maximizing the expected return of scalar rewards. Distributional RL (DRL), in contrast, studies the distribution of returns with the distributional Bellman operator in a Euclidean space, leading to highly flexible choices for utility. This paper establishes robust theoretical foundations for DRL. We prove the contraction property of the Bellman operator even when the reward space is an infinite-dimensional separable Banach space. Furthermore, we demonstrate that the behavior of high- or infinite-dimensional returns can be effective"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.07660","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2024-08-14T16:44:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"bb938693d0a3d6bdd51ee8631dd52c46b6456f4039bed55fee35099b788a8316","abstract_canon_sha256":"59cacb94ea61f1a8de43ffcabbd99490e585e98260e3481fce43060089c2385d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:28.451235Z","signature_b64":"jsGLFEXNA+cNBTpq+/yDFdMqOxd154t6pwZYWF9arraBoyqni8lzKase79W36S9vP4kSiiOz0wlImkj9AqkYBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5dce6b65c383e1f7df78b6e6b864eb5fb07c323a07fa7e6962746d05c8460f09","last_reissued_at":"2026-07-05T08:55:28.450826Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:28.450826Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Off-Policy Reinforcement Learning with High Dimensional Reward","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Dong Neuck Lee, Michael R. Kosorok","submitted_at":"2024-08-14T16:44:56Z","abstract_excerpt":"Conventional off-policy reinforcement learning (RL) focuses on maximizing the expected return of scalar rewards. Distributional RL (DRL), in contrast, studies the distribution of returns with the distributional Bellman operator in a Euclidean space, leading to highly flexible choices for utility. This paper establishes robust theoretical foundations for DRL. We prove the contraction property of the Bellman operator even when the reward space is an infinite-dimensional separable Banach space. Furthermore, we demonstrate that the behavior of high- or infinite-dimensional returns can be effective"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.07660","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.07660/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.07660","created_at":"2026-07-05T08:55:28.450882+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.07660v1","created_at":"2026-07-05T08:55:28.450882+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.07660","created_at":"2026-07-05T08:55:28.450882+00:00"},{"alias_kind":"pith_short_12","alias_value":"LXHGWZODQPQ7","created_at":"2026-07-05T08:55:28.450882+00:00"},{"alias_kind":"pith_short_16","alias_value":"LXHGWZODQPQ7PX3Y","created_at":"2026-07-05T08:55:28.450882+00:00"},{"alias_kind":"pith_short_8","alias_value":"LXHGWZOD","created_at":"2026-07-05T08:55:28.450882+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.11515","citing_title":"AirLLM: Diffusion Policy-based Adaptive LoRA for Remote Fine-Tuning of LLM over the Air","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6","json":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6.json","graph_json":"https://pith.science/api/pith-number/LXHGWZODQPQ7PX3YW3TLQZHLL6/graph.json","events_json":"https://pith.science/api/pith-number/LXHGWZODQPQ7PX3YW3TLQZHLL6/events.json","paper":"https://pith.science/paper/LXHGWZOD"},"agent_actions":{"view_html":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6","download_json":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6.json","view_paper":"https://pith.science/paper/LXHGWZOD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.07660&json=true","fetch_graph":"https://pith.science/api/pith-number/LXHGWZODQPQ7PX3YW3TLQZHLL6/graph.json","fetch_events":"https://pith.science/api/pith-number/LXHGWZODQPQ7PX3YW3TLQZHLL6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6/action/storage_attestation","attest_author":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6/action/author_attestation","sign_citation":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6/action/citation_signature","submit_replication":"https://pith.science/pith/LXHGWZODQPQ7PX3YW3TLQZHLL6/action/replication_record"}},"created_at":"2026-07-05T08:55:28.450882+00:00","updated_at":"2026-07-05T08:55:28.450882+00:00"}