{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:74TUT2VCCPETEIGXVZW22HYXKN","short_pith_number":"pith:74TUT2VC","schema_version":"1.0","canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","source":{"kind":"arxiv","id":"2508.21443","version":1},"attestation_state":"computed","paper":{"title":"Beyond expected value: geometric mean optimization for long-term policy performance in reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Dominik Baumann, Xinyi Sheng","submitted_at":"2025-08-29T09:12:41Z","abstract_excerpt":"Reinforcement learning (RL) algorithms typically optimize the expected cumulative reward, i.e., the expected value of the sum of scalar rewards an agent receives over the course of a trajectory. The expected value averages the performance over an infinite number of trajectories. However, when deploying the agent in the real world, this ensemble average may be uninformative for the performance of individual trajectories. Thus, in many applications, optimizing the long-term performance of individual trajectories might be more desirable. In this work, we propose a novel RL algorithm that combines"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.21443","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-29T09:12:41Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"c77e2770fb37ee059ee77842fc87812a3a190e97c5fda474b6862c405d490c6b","abstract_canon_sha256":"a2006dc6a768df6baaded6c1e3a9a947e971d235ffd6a9b3e82ecc356cdb7e9d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:34.941263Z","signature_b64":"TUss4LPC30+xgt/QjlbQ3O/XP4Ij26UD2F+ndNd2TlUFWShp+cKDMNF7jRKodRQKRbYTpziR1YbzVOo+NQ6gAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff2749eaa213c93220d7ae6dad1f17535572de8cc4b1c5dcb1606678ef6e0877","last_reissued_at":"2026-07-05T12:01:34.940764Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:34.940764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond expected value: geometric mean optimization for long-term policy performance in reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Dominik Baumann, Xinyi Sheng","submitted_at":"2025-08-29T09:12:41Z","abstract_excerpt":"Reinforcement learning (RL) algorithms typically optimize the expected cumulative reward, i.e., the expected value of the sum of scalar rewards an agent receives over the course of a trajectory. The expected value averages the performance over an infinite number of trajectories. However, when deploying the agent in the real world, this ensemble average may be uninformative for the performance of individual trajectories. Thus, in many applications, optimizing the long-term performance of individual trajectories might be more desirable. In this work, we propose a novel RL algorithm that combines"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21443","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.21443/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.21443","created_at":"2026-07-05T12:01:34.940825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.21443v1","created_at":"2026-07-05T12:01:34.940825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21443","created_at":"2026-07-05T12:01:34.940825+00:00"},{"alias_kind":"pith_short_12","alias_value":"74TUT2VCCPET","created_at":"2026-07-05T12:01:34.940825+00:00"},{"alias_kind":"pith_short_16","alias_value":"74TUT2VCCPETEIGX","created_at":"2026-07-05T12:01:34.940825+00:00"},{"alias_kind":"pith_short_8","alias_value":"74TUT2VC","created_at":"2026-07-05T12:01:34.940825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN","json":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN.json","graph_json":"https://pith.science/api/pith-number/74TUT2VCCPETEIGXVZW22HYXKN/graph.json","events_json":"https://pith.science/api/pith-number/74TUT2VCCPETEIGXVZW22HYXKN/events.json","paper":"https://pith.science/paper/74TUT2VC"},"agent_actions":{"view_html":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN","download_json":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN.json","view_paper":"https://pith.science/paper/74TUT2VC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.21443&json=true","fetch_graph":"https://pith.science/api/pith-number/74TUT2VCCPETEIGXVZW22HYXKN/graph.json","fetch_events":"https://pith.science/api/pith-number/74TUT2VCCPETEIGXVZW22HYXKN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/action/storage_attestation","attest_author":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/action/author_attestation","sign_citation":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/action/citation_signature","submit_replication":"https://pith.science/pith/74TUT2VCCPETEIGXVZW22HYXKN/action/replication_record"}},"created_at":"2026-07-05T12:01:34.940825+00:00","updated_at":"2026-07-05T12:01:34.940825+00:00"}