{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:G5FMHXUYLKJCHZM5NSCQCQL7G6","short_pith_number":"pith:G5FMHXUY","schema_version":"1.0","canonical_sha256":"374ac3de985a9223e59d6c8501417f37be2b22c4d523316dd95fae1086a10cc3","source":{"kind":"arxiv","id":"2307.07694","version":3},"attestation_state":"computed","paper":{"title":"Evaluation of Deep Reinforcement Learning Algorithms for Portfolio Optimisation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["q-fin.PM"],"primary_cat":"cs.CE","authors_text":"Chung I Lu","submitted_at":"2023-07-15T03:12:11Z","abstract_excerpt":"We evaluate benchmark deep reinforcement learning algorithms on the task of portfolio optimisation using simulated data. The simulator to generate the data is based on correlated geometric Brownian motion with the Bertsimas-Lo market impact model. Using the Kelly criterion (log utility) as the objective, we can analytically derive the optimal policy without market impact as an upper bound to measure performance when including market impact. We find that the off-policy algorithms DDPG, TD3 and SAC are unable to learn the right $Q$-function due to the noisy rewards and therefore perform poorly. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.07694","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CE","submitted_at":"2023-07-15T03:12:11Z","cross_cats_sorted":["q-fin.PM"],"title_canon_sha256":"c7f8bcc43cfa4f4ca1b8aef74b15c6e9e2f390f1292ac05b607b94227e1b791d","abstract_canon_sha256":"3e6aff162194a37fe33956876bca4c85a602bfaff90551f17a36f5ac1375c8fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:02.228664Z","signature_b64":"mVQnZfb7cPhx9LDtCV8ukm4i3BjSiPL4dcyAy6BgdRfYILaBU7xFJfkNEb/0v7y4UMlztB59W7yZy4wS9ALVDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"374ac3de985a9223e59d6c8501417f37be2b22c4d523316dd95fae1086a10cc3","last_reissued_at":"2026-07-05T11:49:02.228155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:02.228155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of Deep Reinforcement Learning Algorithms for Portfolio Optimisation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["q-fin.PM"],"primary_cat":"cs.CE","authors_text":"Chung I Lu","submitted_at":"2023-07-15T03:12:11Z","abstract_excerpt":"We evaluate benchmark deep reinforcement learning algorithms on the task of portfolio optimisation using simulated data. The simulator to generate the data is based on correlated geometric Brownian motion with the Bertsimas-Lo market impact model. Using the Kelly criterion (log utility) as the objective, we can analytically derive the optimal policy without market impact as an upper bound to measure performance when including market impact. We find that the off-policy algorithms DDPG, TD3 and SAC are unable to learn the right $Q$-function due to the noisy rewards and therefore perform poorly. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.07694","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.07694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.07694","created_at":"2026-07-05T11:49:02.228211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.07694v3","created_at":"2026-07-05T11:49:02.228211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.07694","created_at":"2026-07-05T11:49:02.228211+00:00"},{"alias_kind":"pith_short_12","alias_value":"G5FMHXUYLKJC","created_at":"2026-07-05T11:49:02.228211+00:00"},{"alias_kind":"pith_short_16","alias_value":"G5FMHXUYLKJCHZM5","created_at":"2026-07-05T11:49:02.228211+00:00"},{"alias_kind":"pith_short_8","alias_value":"G5FMHXUY","created_at":"2026-07-05T11:49:02.228211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02619","citing_title":"Regret-Optimized Portfolio Enhancement through Deep Reinforcement Learning and Future Looking Rewards","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6","json":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6.json","graph_json":"https://pith.science/api/pith-number/G5FMHXUYLKJCHZM5NSCQCQL7G6/graph.json","events_json":"https://pith.science/api/pith-number/G5FMHXUYLKJCHZM5NSCQCQL7G6/events.json","paper":"https://pith.science/paper/G5FMHXUY"},"agent_actions":{"view_html":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6","download_json":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6.json","view_paper":"https://pith.science/paper/G5FMHXUY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.07694&json=true","fetch_graph":"https://pith.science/api/pith-number/G5FMHXUYLKJCHZM5NSCQCQL7G6/graph.json","fetch_events":"https://pith.science/api/pith-number/G5FMHXUYLKJCHZM5NSCQCQL7G6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6/action/storage_attestation","attest_author":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6/action/author_attestation","sign_citation":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6/action/citation_signature","submit_replication":"https://pith.science/pith/G5FMHXUYLKJCHZM5NSCQCQL7G6/action/replication_record"}},"created_at":"2026-07-05T11:49:02.228211+00:00","updated_at":"2026-07-05T11:49:02.228211+00:00"}