{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PQXKKBJQXIOQQUNJ7IYEB2AHMD","short_pith_number":"pith:PQXKKBJQ","schema_version":"1.0","canonical_sha256":"7c2ea50530ba1d0851a9fa3040e80760e3b1689804f6a4eecda0c93e7f5c07f1","source":{"kind":"arxiv","id":"2201.01874","version":1},"attestation_state":"computed","paper":{"title":"Combining Reinforcement Learning and Inverse Reinforcement Learning for Asset Allocation Recommendations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","q-fin.CP","q-fin.PM"],"primary_cat":"cs.LG","authors_text":"Igor Halperin, Jiayu Liu, Xiao Zhang","submitted_at":"2022-01-06T00:08:17Z","abstract_excerpt":"We suggest a simple practical method to combine the human and artificial intelligence to both learn best investment practices of fund managers, and provide recommendations to improve them. Our approach is based on a combination of Inverse Reinforcement Learning (IRL) and RL. First, the IRL component learns the intent of fund managers as suggested by their trading history, and recovers their implied reward function. At the second step, this reward function is used by a direct RL algorithm to optimize asset allocation decisions. We show that our method is able to improve over the performance of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.01874","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-01-06T00:08:17Z","cross_cats_sorted":["cs.AI","q-fin.CP","q-fin.PM"],"title_canon_sha256":"cb916a25b5e558d55efc9527d94ab71b9add6855dff3e58b613db7339833fb6c","abstract_canon_sha256":"959e280ccb0fb3451f3290a3204f854087162444995d1f553fd7ee582705c043"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:46:27.627043Z","signature_b64":"qXXmTaWdCTpC3dmpLZVh+/ZltRdw+NrSfQHJB5uPQUzSYqLAL1OdPHUfBkGz8Bf9UgSWcI9/VfcpJNyRjDMdDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c2ea50530ba1d0851a9fa3040e80760e3b1689804f6a4eecda0c93e7f5c07f1","last_reissued_at":"2026-07-05T03:46:27.626625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:46:27.626625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Combining Reinforcement Learning and Inverse Reinforcement Learning for Asset Allocation Recommendations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","q-fin.CP","q-fin.PM"],"primary_cat":"cs.LG","authors_text":"Igor Halperin, Jiayu Liu, Xiao Zhang","submitted_at":"2022-01-06T00:08:17Z","abstract_excerpt":"We suggest a simple practical method to combine the human and artificial intelligence to both learn best investment practices of fund managers, and provide recommendations to improve them. Our approach is based on a combination of Inverse Reinforcement Learning (IRL) and RL. First, the IRL component learns the intent of fund managers as suggested by their trading history, and recovers their implied reward function. At the second step, this reward function is used by a direct RL algorithm to optimize asset allocation decisions. We show that our method is able to improve over the performance of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.01874","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.01874/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.01874","created_at":"2026-07-05T03:46:27.626688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.01874v1","created_at":"2026-07-05T03:46:27.626688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.01874","created_at":"2026-07-05T03:46:27.626688+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQXKKBJQXIOQ","created_at":"2026-07-05T03:46:27.626688+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQXKKBJQXIOQQUNJ","created_at":"2026-07-05T03:46:27.626688+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQXKKBJQ","created_at":"2026-07-05T03:46:27.626688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02619","citing_title":"Regret-Optimized Portfolio Enhancement through Deep Reinforcement Learning and Future Looking Rewards","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD","json":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD.json","graph_json":"https://pith.science/api/pith-number/PQXKKBJQXIOQQUNJ7IYEB2AHMD/graph.json","events_json":"https://pith.science/api/pith-number/PQXKKBJQXIOQQUNJ7IYEB2AHMD/events.json","paper":"https://pith.science/paper/PQXKKBJQ"},"agent_actions":{"view_html":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD","download_json":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD.json","view_paper":"https://pith.science/paper/PQXKKBJQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.01874&json=true","fetch_graph":"https://pith.science/api/pith-number/PQXKKBJQXIOQQUNJ7IYEB2AHMD/graph.json","fetch_events":"https://pith.science/api/pith-number/PQXKKBJQXIOQQUNJ7IYEB2AHMD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD/action/storage_attestation","attest_author":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD/action/author_attestation","sign_citation":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD/action/citation_signature","submit_replication":"https://pith.science/pith/PQXKKBJQXIOQQUNJ7IYEB2AHMD/action/replication_record"}},"created_at":"2026-07-05T03:46:27.626688+00:00","updated_at":"2026-07-05T03:46:27.626688+00:00"}