{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:74TUT2VCCPETEIGXVZW22HYXKN","short_pith_number":"pith:74TUT2VC","canonical_record":{"source":{"id":"2508.21443","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-29T09:12:41Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"c77e2770fb37ee059ee77842fc87812a3a190e97c5fda474b6862c405d490c6b","abstract_canon_sha256":"a2006dc6a768df6baaded6c1e3a9a947e971d235ffd6a9b3e82ecc356cdb7e9d"},"schema_version":"1.0"},"canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","source":{"kind":"arxiv","id":"2508.21443","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.21443","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"arxiv_version","alias_value":"2508.21443v1","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21443","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"pith_short_12","alias_value":"74TUT2VCCPET","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"pith_short_16","alias_value":"74TUT2VCCPETEIGX","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"pith_short_8","alias_value":"74TUT2VC","created_at":"2026-07-05T12:01:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:74TUT2VCCPETEIGXVZW22HYXKN","target":"record","payload":{"canonical_record":{"source":{"id":"2508.21443","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-29T09:12:41Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"c77e2770fb37ee059ee77842fc87812a3a190e97c5fda474b6862c405d490c6b","abstract_canon_sha256":"a2006dc6a768df6baaded6c1e3a9a947e971d235ffd6a9b3e82ecc356cdb7e9d"},"schema_version":"1.0"},"canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:34.941263Z","signature_b64":"TUss4LPC30+xgt/QjlbQ3O/XP4Ij26UD2F+ndNd2TlUFWShp+cKDMNF7jRKodRQKRbYTpziR1YbzVOo+NQ6gAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","last_reissued_at":"2026-07-05T12:01:34.940764Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:34.940764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2508.21443","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:01:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qOUfG1H3Y0l6h1iQynUwIhOM8cV0T2MwhGQcuxgFb12AVfCYW8dNBpPhIl/h4hAcZEWSDk7Zf6HpDH3105FEDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T17:46:09.966210Z"},"content_sha256":"9ee4b9eb141932ed8530f0cb02e373cc877f3bcbf586ef652f9e3eaca506746f","schema_version":"1.0","event_id":"sha256:9ee4b9eb141932ed8530f0cb02e373cc877f3bcbf586ef652f9e3eaca506746f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:74TUT2VCCPETEIGXVZW22HYXKN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond expected value: geometric mean optimization for long-term policy performance in reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Dominik Baumann, Xinyi Sheng","submitted_at":"2025-08-29T09:12:41Z","abstract_excerpt":"Reinforcement learning (RL) algorithms typically optimize the expected cumulative reward, i.e., the expected value of the sum of scalar rewards an agent receives over the course of a trajectory. The expected value averages the performance over an infinite number of trajectories. However, when deploying the agent in the real world, this ensemble average may be uninformative for the performance of individual trajectories. Thus, in many applications, optimizing the long-term performance of individual trajectories might be more desirable. In this work, we propose a novel RL algorithm that combines"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21443","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.21443/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:01:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uWcOY4pjkB7h8lBEgBGgf8JTUJ7xpE3st8fJnSsIjH0L+qe6v0wJqD8Fh2se6zdM7CorvfNUuo3POP3Ag7CNCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T17:46:09.966702Z"},"content_sha256":"97c75b36af3e09246b6be244c4ea725f31068fc05f1baef64f1f854f97672d8f","schema_version":"1.0","event_id":"sha256:97c75b36af3e09246b6be244c4ea725f31068fc05f1baef64f1f854f97672d8f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/bundle.json","state_url":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/74TUT2VCCPETEIGXVZW22HYXKN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-16T17:46:09Z","links":{"resolver":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN","bundle":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/bundle.json","state":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/74TUT2VCCPETEIGXVZW22HYXKN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:74TUT2VCCPETEIGXVZW22HYXKN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a2006dc6a768df6baaded6c1e3a9a947e971d235ffd6a9b3e82ecc356cdb7e9d","cross_cats_sorted":["cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-29T09:12:41Z","title_canon_sha256":"c77e2770fb37ee059ee77842fc87812a3a190e97c5fda474b6862c405d490c6b"},"schema_version":"1.0","source":{"id":"2508.21443","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.21443","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"arxiv_version","alias_value":"2508.21443v1","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21443","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"pith_short_12","alias_value":"74TUT2VCCPET","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"pith_short_16","alias_value":"74TUT2VCCPETEIGX","created_at":"2026-07-05T12:01:34Z"},{"alias_kind":"pith_short_8","alias_value":"74TUT2VC","created_at":"2026-07-05T12:01:34Z"}],"graph_snapshots":[{"event_id":"sha256:97c75b36af3e09246b6be244c4ea725f31068fc05f1baef64f1f854f97672d8f","target":"graph","created_at":"2026-07-05T12:01:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2508.21443/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) algorithms typically optimize the expected cumulative reward, i.e., the expected value of the sum of scalar rewards an agent receives over the course of a trajectory. The expected value averages the performance over an infinite number of trajectories. However, when deploying the agent in the real world, this ensemble average may be uninformative for the performance of individual trajectories. Thus, in many applications, optimizing the long-term performance of individual trajectories might be more desirable. In this work, we propose a novel RL algorithm that combines","authors_text":"Dominik Baumann, Xinyi Sheng","cross_cats":["cs.SY","eess.SY"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-29T09:12:41Z","title":"Beyond expected value: geometric mean optimization for long-term policy performance in reinforcement learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21443","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9ee4b9eb141932ed8530f0cb02e373cc877f3bcbf586ef652f9e3eaca506746f","target":"record","created_at":"2026-07-05T12:01:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a2006dc6a768df6baaded6c1e3a9a947e971d235ffd6a9b3e82ecc356cdb7e9d","cross_cats_sorted":["cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-29T09:12:41Z","title_canon_sha256":"c77e2770fb37ee059ee77842fc87812a3a190e97c5fda474b6862c405d490c6b"},"schema_version":"1.0","source":{"id":"2508.21443","kind":"arxiv","version":1}},"canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","first_computed_at":"2026-07-05T12:01:34.940764Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:01:34.940764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"TUss4LPC30+xgt/QjlbQ3O/XP4Ij26UD2F+ndNd2TlUFWShp+cKDMNF7jRKodRQKRbYTpziR1YbzVOo+NQ6gAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T12:01:34.941263Z","signed_message":"canonical_sha256_bytes"},"source_id":"2508.21443","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9ee4b9eb141932ed8530f0cb02e373cc877f3bcbf586ef652f9e3eaca506746f","sha256:97c75b36af3e09246b6be244c4ea725f31068fc05f1baef64f1f854f97672d8f"],"state_sha256":"fafe1c9601442c1d447e03ec65db2baefd219a3612c4841fa5c8f665481d36bc"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uwYUYT3gBIsgW6cQGVxjvG7ZYqrA22eob6dKmEV27fZ/wq1jOTFaQ1kaURf7DaoYvN0t9Y+VgSXMo8yONc8ODw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-16T17:46:09.971900Z","bundle_sha256":"7c262f45a5d4c0c73947dfda7087dbecf66bdfbac6e58b64a32fb4288b5315a1"}}