{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AUOQFTAGGAUMVCX4AR6WYRRKAT","short_pith_number":"pith:AUOQFTAG","schema_version":"1.0","canonical_sha256":"051d02cc063028ca8afc047d6c462a04fa0cadad9fa508ce3474d564fc93cc6e","source":{"kind":"arxiv","id":"2410.13493","version":1},"attestation_state":"computed","paper":{"title":"Deep Reinforcement Learning for Online Optimal Execution Strategies","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alessandro Micheli, M\\'elodie Monod","submitted_at":"2024-10-17T12:38:08Z","abstract_excerpt":"This paper tackles the challenge of learning non-Markovian optimal execution strategies in dynamic financial markets. We introduce a novel actor-critic algorithm based on Deep Deterministic Policy Gradient (DDPG) to address this issue, with a focus on transient price impact modeled by a general decay kernel. Through numerical experiments with various decay kernels, we show that our algorithm successfully approximates the optimal execution strategy. Additionally, the proposed algorithm demonstrates adaptability to evolving market conditions, where parameters fluctuate over time. Our findings al"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13493","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-17T12:38:08Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"a0caa7de1654f9993ed8db7e1d9f9329ea30553547a2ebc0b07ecef401075952","abstract_canon_sha256":"35c0261ea20406ac545d9b2cc9f83f4b9b35a08d932bd7abac5f4586d5523797"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:01.832220Z","signature_b64":"jsmiuXmoE6WNKS2ugPtAl3vnWNl2mdCkTfsBvtX+y3OmNi3cSs/KJttr+yFmQD1CjF3axSazyjHWc1L8kcQjDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"051d02cc063028ca8afc047d6c462a04fa0cadad9fa508ce3474d564fc93cc6e","last_reissued_at":"2026-07-05T09:22:01.831734Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:01.831734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Reinforcement Learning for Online Optimal Execution Strategies","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alessandro Micheli, M\\'elodie Monod","submitted_at":"2024-10-17T12:38:08Z","abstract_excerpt":"This paper tackles the challenge of learning non-Markovian optimal execution strategies in dynamic financial markets. We introduce a novel actor-critic algorithm based on Deep Deterministic Policy Gradient (DDPG) to address this issue, with a focus on transient price impact modeled by a general decay kernel. Through numerical experiments with various decay kernels, we show that our algorithm successfully approximates the optimal execution strategy. Additionally, the proposed algorithm demonstrates adaptability to evolving market conditions, where parameters fluctuate over time. Our findings al"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13493","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13493/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13493","created_at":"2026-07-05T09:22:01.831805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13493v1","created_at":"2026-07-05T09:22:01.831805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13493","created_at":"2026-07-05T09:22:01.831805+00:00"},{"alias_kind":"pith_short_12","alias_value":"AUOQFTAGGAUM","created_at":"2026-07-05T09:22:01.831805+00:00"},{"alias_kind":"pith_short_16","alias_value":"AUOQFTAGGAUMVCX4","created_at":"2026-07-05T09:22:01.831805+00:00"},{"alias_kind":"pith_short_8","alias_value":"AUOQFTAG","created_at":"2026-07-05T09:22:01.831805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06121","citing_title":"Can Reinforcement Learning Efficiently Discover Price Manipulation?","ref_index":66,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08379","citing_title":"TT-DAC-PS: Twin-Target Deterministic Actor-Critic with Policy Smoothing for Optimal Trade Execution","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20348","citing_title":"Memory-Induced Supra-Competitive Outcomes Between Deep Reinforcement Learning Agents in Optimal Trade Execution","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT","json":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT.json","graph_json":"https://pith.science/api/pith-number/AUOQFTAGGAUMVCX4AR6WYRRKAT/graph.json","events_json":"https://pith.science/api/pith-number/AUOQFTAGGAUMVCX4AR6WYRRKAT/events.json","paper":"https://pith.science/paper/AUOQFTAG"},"agent_actions":{"view_html":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT","download_json":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT.json","view_paper":"https://pith.science/paper/AUOQFTAG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13493&json=true","fetch_graph":"https://pith.science/api/pith-number/AUOQFTAGGAUMVCX4AR6WYRRKAT/graph.json","fetch_events":"https://pith.science/api/pith-number/AUOQFTAGGAUMVCX4AR6WYRRKAT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT/action/storage_attestation","attest_author":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT/action/author_attestation","sign_citation":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT/action/citation_signature","submit_replication":"https://pith.science/pith/AUOQFTAGGAUMVCX4AR6WYRRKAT/action/replication_record"}},"created_at":"2026-07-05T09:22:01.831805+00:00","updated_at":"2026-07-05T09:22:01.831805+00:00"}