{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KLMNF6K3RD3ZRBSQZ6W6XH3AUF","short_pith_number":"pith:KLMNF6K3","schema_version":"1.0","canonical_sha256":"52d8d2f95b88f7988650cfadeb9f60a143fa8b83d29a9afbd449050479ce90d0","source":{"kind":"arxiv","id":"2405.13609","version":3},"attestation_state":"computed","paper":{"title":"Tackling Decision Processes with Non-Cumulative Objectives using Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["q-fin.CP","quant-ph"],"primary_cat":"cs.LG","authors_text":"Florian Marquardt, Jan Olle, Maximilian N\\\"agele, Remmy Zen, Thomas F\\\"osel","submitted_at":"2024-05-22T13:01:37Z","abstract_excerpt":"Markov decision processes (MDPs) are used to model a wide variety of applications ranging from game playing over robotics to finance. Their optimal policy typically maximizes the expected sum of rewards given at each step of the decision process. However, a large class of problems does not fit straightforwardly into this framework: Non-cumulative Markov decision processes (NCMDPs), where instead of the expected sum of rewards, the expected value of an arbitrary function of the rewards is maximized. Example functions include the maximum of the rewards or their mean divided by their standard dev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13609","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-22T13:01:37Z","cross_cats_sorted":["q-fin.CP","quant-ph"],"title_canon_sha256":"3ef008f933c7d501094767d4a63850edaabefdff8bd916cf3c452f5dc05f7ec0","abstract_canon_sha256":"adc9f4c8813be4308fa616e9277f1b1b4a4bafe4dec67f8d6e933cbd62360a32"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:57.579670Z","signature_b64":"fyhh0TcyRhfst/BwCB7fmbJygg0VH83/PWwFBY1GUm50B1kwHJ51w6sKsT0DZhMyT4JsspXqNJ1J1C/RIHY5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52d8d2f95b88f7988650cfadeb9f60a143fa8b83d29a9afbd449050479ce90d0","last_reissued_at":"2026-07-05T11:07:57.579142Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:57.579142Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tackling Decision Processes with Non-Cumulative Objectives using Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["q-fin.CP","quant-ph"],"primary_cat":"cs.LG","authors_text":"Florian Marquardt, Jan Olle, Maximilian N\\\"agele, Remmy Zen, Thomas F\\\"osel","submitted_at":"2024-05-22T13:01:37Z","abstract_excerpt":"Markov decision processes (MDPs) are used to model a wide variety of applications ranging from game playing over robotics to finance. Their optimal policy typically maximizes the expected sum of rewards given at each step of the decision process. However, a large class of problems does not fit straightforwardly into this framework: Non-cumulative Markov decision processes (NCMDPs), where instead of the expected sum of rewards, the expected value of an arbitrary function of the rewards is maximized. Example functions include the maximum of the rewards or their mean divided by their standard dev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13609","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13609/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13609","created_at":"2026-07-05T11:07:57.579198+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13609v3","created_at":"2026-07-05T11:07:57.579198+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13609","created_at":"2026-07-05T11:07:57.579198+00:00"},{"alias_kind":"pith_short_12","alias_value":"KLMNF6K3RD3Z","created_at":"2026-07-05T11:07:57.579198+00:00"},{"alias_kind":"pith_short_16","alias_value":"KLMNF6K3RD3ZRBSQ","created_at":"2026-07-05T11:07:57.579198+00:00"},{"alias_kind":"pith_short_8","alias_value":"KLMNF6K3","created_at":"2026-07-05T11:07:57.579198+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.08537","citing_title":"Recursive Reward Aggregation","ref_index":60,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF","json":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF.json","graph_json":"https://pith.science/api/pith-number/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/graph.json","events_json":"https://pith.science/api/pith-number/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/events.json","paper":"https://pith.science/paper/KLMNF6K3"},"agent_actions":{"view_html":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF","download_json":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF.json","view_paper":"https://pith.science/paper/KLMNF6K3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13609&json=true","fetch_graph":"https://pith.science/api/pith-number/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/graph.json","fetch_events":"https://pith.science/api/pith-number/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/action/storage_attestation","attest_author":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/action/author_attestation","sign_citation":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/action/citation_signature","submit_replication":"https://pith.science/pith/KLMNF6K3RD3ZRBSQZ6W6XH3AUF/action/replication_record"}},"created_at":"2026-07-05T11:07:57.579198+00:00","updated_at":"2026-07-05T11:07:57.579198+00:00"}