{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M43TSJOAANR3EIRTYNU3X5VCVH","short_pith_number":"pith:M43TSJOA","schema_version":"1.0","canonical_sha256":"67373925c00363b22233c369bbf6a2a9fb33a6b8d8fef749f288119a14bbe8dd","source":{"kind":"arxiv","id":"2507.20263","version":1},"attestation_state":"computed","paper":{"title":"Learning from Expert Factors: Trajectory-level Reward Shaping for Formulaic Alpha Mining","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-fin.PM"],"primary_cat":"cs.LG","authors_text":"Chengxi Zhang, Chenkai Wang, Junjie Zhao, Peng Yang","submitted_at":"2025-07-27T13:14:48Z","abstract_excerpt":"Reinforcement learning (RL) has successfully automated the complex process of mining formulaic alpha factors, for creating interpretable and profitable investment strategies. However, existing methods are hampered by the sparse rewards given the underlying Markov Decision Process. This inefficiency limits the exploration of the vast symbolic search space and destabilizes the training process. To address this, Trajectory-level Reward Shaping (TLRS), a novel reward shaping method, is proposed. TLRS provides dense, intermediate rewards by measuring the subsequence-level similarity between partial"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.20263","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-27T13:14:48Z","cross_cats_sorted":["cs.AI","q-fin.PM"],"title_canon_sha256":"912acb27b0b34d4757400b3cf1f0c163456c627c5f69a57a9d4cb0771dd37f7a","abstract_canon_sha256":"73dab0361aa2bd0a887a1f30dab63b1d08e6d1ba1cc8f8b54a0cacee21353185"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:16.018902Z","signature_b64":"nyskcMdZbCpM/LEUEECAfVF3RiY3ksIgk2EojhbPzDdpFji77ZyswIQtUdBJ77NkFtaLjw2v1wLQYOW/E8NLDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67373925c00363b22233c369bbf6a2a9fb33a6b8d8fef749f288119a14bbe8dd","last_reissued_at":"2026-07-05T11:44:16.018467Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:16.018467Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Expert Factors: Trajectory-level Reward Shaping for Formulaic Alpha Mining","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-fin.PM"],"primary_cat":"cs.LG","authors_text":"Chengxi Zhang, Chenkai Wang, Junjie Zhao, Peng Yang","submitted_at":"2025-07-27T13:14:48Z","abstract_excerpt":"Reinforcement learning (RL) has successfully automated the complex process of mining formulaic alpha factors, for creating interpretable and profitable investment strategies. However, existing methods are hampered by the sparse rewards given the underlying Markov Decision Process. This inefficiency limits the exploration of the vast symbolic search space and destabilizes the training process. To address this, Trajectory-level Reward Shaping (TLRS), a novel reward shaping method, is proposed. TLRS provides dense, intermediate rewards by measuring the subsequence-level similarity between partial"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.20263","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.20263/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.20263","created_at":"2026-07-05T11:44:16.018527+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.20263v1","created_at":"2026-07-05T11:44:16.018527+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.20263","created_at":"2026-07-05T11:44:16.018527+00:00"},{"alias_kind":"pith_short_12","alias_value":"M43TSJOAANR3","created_at":"2026-07-05T11:44:16.018527+00:00"},{"alias_kind":"pith_short_16","alias_value":"M43TSJOAANR3EIRT","created_at":"2026-07-05T11:44:16.018527+00:00"},{"alias_kind":"pith_short_8","alias_value":"M43TSJOA","created_at":"2026-07-05T11:44:16.018527+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.25055","citing_title":"AlphaSAGE: Structure-Aware Alpha Mining via GFlowNets for Robust Exploration","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH","json":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH.json","graph_json":"https://pith.science/api/pith-number/M43TSJOAANR3EIRTYNU3X5VCVH/graph.json","events_json":"https://pith.science/api/pith-number/M43TSJOAANR3EIRTYNU3X5VCVH/events.json","paper":"https://pith.science/paper/M43TSJOA"},"agent_actions":{"view_html":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH","download_json":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH.json","view_paper":"https://pith.science/paper/M43TSJOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.20263&json=true","fetch_graph":"https://pith.science/api/pith-number/M43TSJOAANR3EIRTYNU3X5VCVH/graph.json","fetch_events":"https://pith.science/api/pith-number/M43TSJOAANR3EIRTYNU3X5VCVH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH/action/storage_attestation","attest_author":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH/action/author_attestation","sign_citation":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH/action/citation_signature","submit_replication":"https://pith.science/pith/M43TSJOAANR3EIRTYNU3X5VCVH/action/replication_record"}},"created_at":"2026-07-05T11:44:16.018527+00:00","updated_at":"2026-07-05T11:44:16.018527+00:00"}