{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:YSSCO5MIGGNMUGFMAAZF74AUCM","short_pith_number":"pith:YSSCO5MI","canonical_record":{"source":{"id":"2407.13146","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T04:18:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3999079349135068f83ece080d6c96fb95b32a56595037bcd5af03d69e01aa17","abstract_canon_sha256":"e17fc8fbccaf3b83a47be4060af3ddab3c881f276b6f9d43b62451304afab531"},"schema_version":"1.0"},"canonical_sha256":"c4a4277588319aca18ac00325ff01413306a75319e1d8845defaaefc24e5dbee","source":{"kind":"arxiv","id":"2407.13146","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.13146","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"arxiv_version","alias_value":"2407.13146v2","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13146","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"pith_short_12","alias_value":"YSSCO5MIGGNM","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"pith_short_16","alias_value":"YSSCO5MIGGNMUGFM","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"pith_short_8","alias_value":"YSSCO5MI","created_at":"2026-07-05T08:45:52Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:YSSCO5MIGGNMUGFMAAZF74AUCM","target":"record","payload":{"canonical_record":{"source":{"id":"2407.13146","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T04:18:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3999079349135068f83ece080d6c96fb95b32a56595037bcd5af03d69e01aa17","abstract_canon_sha256":"e17fc8fbccaf3b83a47be4060af3ddab3c881f276b6f9d43b62451304afab531"},"schema_version":"1.0"},"canonical_sha256":"c4a4277588319aca18ac00325ff01413306a75319e1d8845defaaefc24e5dbee","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:52.960708Z","signature_b64":"vDm/2YjNNN16ksxwXHjjKIcHWU0tMmeYQ9a3AuaoDFwMIsUH7RLIdVsVPuP68kxEll9rfynUUj0D0FTpmkQoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4a4277588319aca18ac00325ff01413306a75319e1d8845defaaefc24e5dbee","last_reissued_at":"2026-07-05T08:45:52.960250Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:52.960250Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2407.13146","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:45:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2I/lIKALXXw+KcPIZrnz9jRrs0ozUvOBDvqp/aQAP1ZC6cmMDbZn1Y4WNkpOr25LGRH2STXwgPMwOaPAS17NBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T20:46:16.647287Z"},"content_sha256":"989c907a3d60ca9995f85fe8759fd66cb1986bdd7cf0f13f434a333c09bc1bb1","schema_version":"1.0","event_id":"sha256:989c907a3d60ca9995f85fe8759fd66cb1986bdd7cf0f13f434a333c09bc1bb1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:YSSCO5MIGGNMUGFMAAZF74AUCM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"PG-Rainbow: Using Distributional Reinforcement Learning in Policy Gradient Methods","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jeewoo Lee, KangJun Lee, WooJae Jeon","submitted_at":"2024-07-18T04:18:52Z","abstract_excerpt":"This paper introduces PG-Rainbow, a novel algorithm that incorporates a distributional reinforcement learning framework with a policy gradient algorithm. Existing policy gradient methods are sample inefficient and rely on the mean of returns when calculating the state-action value function, neglecting the distributional nature of returns in reinforcement learning tasks. To address this issue, we use an Implicit Quantile Network that provides the quantile information of the distribution of rewards to the critic network of the Proximal Policy Optimization algorithm. We show empirical results tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13146","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13146/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:45:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"g1EKr+TsRNX6Wyei9C6qKpxIkrIsAkAUSleuonmuZkHIMZ7N+cK5qA2fZRzrisMopVgRymTUcd+Na6p8nTFDCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T20:46:16.647628Z"},"content_sha256":"bb52da530b69483f22caf2153bcf57da142525faddd8b2a20c9683f340187922","schema_version":"1.0","event_id":"sha256:bb52da530b69483f22caf2153bcf57da142525faddd8b2a20c9683f340187922"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YSSCO5MIGGNMUGFMAAZF74AUCM/bundle.json","state_url":"https://pith.science/pith/YSSCO5MIGGNMUGFMAAZF74AUCM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YSSCO5MIGGNMUGFMAAZF74AUCM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T20:46:16Z","links":{"resolver":"https://pith.science/pith/YSSCO5MIGGNMUGFMAAZF74AUCM","bundle":"https://pith.science/pith/YSSCO5MIGGNMUGFMAAZF74AUCM/bundle.json","state":"https://pith.science/pith/YSSCO5MIGGNMUGFMAAZF74AUCM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YSSCO5MIGGNMUGFMAAZF74AUCM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:YSSCO5MIGGNMUGFMAAZF74AUCM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e17fc8fbccaf3b83a47be4060af3ddab3c881f276b6f9d43b62451304afab531","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T04:18:52Z","title_canon_sha256":"3999079349135068f83ece080d6c96fb95b32a56595037bcd5af03d69e01aa17"},"schema_version":"1.0","source":{"id":"2407.13146","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.13146","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"arxiv_version","alias_value":"2407.13146v2","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13146","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"pith_short_12","alias_value":"YSSCO5MIGGNM","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"pith_short_16","alias_value":"YSSCO5MIGGNMUGFM","created_at":"2026-07-05T08:45:52Z"},{"alias_kind":"pith_short_8","alias_value":"YSSCO5MI","created_at":"2026-07-05T08:45:52Z"}],"graph_snapshots":[{"event_id":"sha256:bb52da530b69483f22caf2153bcf57da142525faddd8b2a20c9683f340187922","target":"graph","created_at":"2026-07-05T08:45:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2407.13146/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper introduces PG-Rainbow, a novel algorithm that incorporates a distributional reinforcement learning framework with a policy gradient algorithm. Existing policy gradient methods are sample inefficient and rely on the mean of returns when calculating the state-action value function, neglecting the distributional nature of returns in reinforcement learning tasks. To address this issue, we use an Implicit Quantile Network that provides the quantile information of the distribution of rewards to the critic network of the Proximal Policy Optimization algorithm. We show empirical results tha","authors_text":"Jeewoo Lee, KangJun Lee, WooJae Jeon","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T04:18:52Z","title":"PG-Rainbow: Using Distributional Reinforcement Learning in Policy Gradient Methods"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13146","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:989c907a3d60ca9995f85fe8759fd66cb1986bdd7cf0f13f434a333c09bc1bb1","target":"record","created_at":"2026-07-05T08:45:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e17fc8fbccaf3b83a47be4060af3ddab3c881f276b6f9d43b62451304afab531","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T04:18:52Z","title_canon_sha256":"3999079349135068f83ece080d6c96fb95b32a56595037bcd5af03d69e01aa17"},"schema_version":"1.0","source":{"id":"2407.13146","kind":"arxiv","version":2}},"canonical_sha256":"c4a4277588319aca18ac00325ff01413306a75319e1d8845defaaefc24e5dbee","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c4a4277588319aca18ac00325ff01413306a75319e1d8845defaaefc24e5dbee","first_computed_at":"2026-07-05T08:45:52.960250Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:45:52.960250Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"vDm/2YjNNN16ksxwXHjjKIcHWU0tMmeYQ9a3AuaoDFwMIsUH7RLIdVsVPuP68kxEll9rfynUUj0D0FTpmkQoAg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:45:52.960708Z","signed_message":"canonical_sha256_bytes"},"source_id":"2407.13146","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:989c907a3d60ca9995f85fe8759fd66cb1986bdd7cf0f13f434a333c09bc1bb1","sha256:bb52da530b69483f22caf2153bcf57da142525faddd8b2a20c9683f340187922"],"state_sha256":"9c5c4158f71249ecee3614a8038cbef7c705941d91d8a7a8d7b0eac5558608ff"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7VLX+pXkPHrSU6NHMN2fsQjDHGbRdCqTdkSpePeovlMb19aK8whh+Pn272vyIm6KqmrzghzC+5n+/sz/W6iGBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T20:46:16.651187Z","bundle_sha256":"30b6db073e47a6b148b20300166a6a6603da9cf0dc94b07daace5b7d7a4b3a18"}}