{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3BHLKV43773L5FXOD2SVJNU3BB","short_pith_number":"pith:3BHLKV43","schema_version":"1.0","canonical_sha256":"d84eb5579bfff6be96ee1ea554b69b08579ce956398c31ef53c68e76c1ab35e7","source":{"kind":"arxiv","id":"2506.23090","version":2},"attestation_state":"computed","paper":{"title":"Multi-task Offline Reinforcement Learning for Online Advertising in Recommender Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Bo Li, Bo Zheng, Chi Zhang, Hongzhi Yin, Langming Liu, Wanyu Wang, Wenbo Su, Xiangyu Zhao, Xuetao Wei","submitted_at":"2025-06-29T05:05:13Z","abstract_excerpt":"Online advertising in recommendation platforms has gained significant attention, with a predominant focus on channel recommendation and budget allocation strategies. However, current offline reinforcement learning (RL) methods face substantial challenges when applied to sparse advertising scenarios, primarily due to severe overestimation, distributional shifts, and overlooking budget constraints. To address these issues, we propose MTORL, a novel multi-task offline RL model that targets two key objectives. First, we establish a Markov Decision Process (MDP) framework specific to the nuances of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.23090","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2025-06-29T05:05:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"597e9ebc3cb39749f2fdeb18e5eb55f1b7f047a2dfbaf60f2b7ac20e020ad0e3","abstract_canon_sha256":"28e76aa3761c1a8176d25b4654ce774e33027f7a0f1b5aef3ee393a9509ffd94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:10.150543Z","signature_b64":"C4uQWr4xdD0laoTVehNHYs1HzntL3XIccHR66SwV+iMHU6wki66R0OeD469D2PqIaCqxtc0Lglsb2X+julv9Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d84eb5579bfff6be96ee1ea554b69b08579ce956398c31ef53c68e76c1ab35e7","last_reissued_at":"2026-07-05T11:34:10.150086Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:10.150086Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-task Offline Reinforcement Learning for Online Advertising in Recommender Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Bo Li, Bo Zheng, Chi Zhang, Hongzhi Yin, Langming Liu, Wanyu Wang, Wenbo Su, Xiangyu Zhao, Xuetao Wei","submitted_at":"2025-06-29T05:05:13Z","abstract_excerpt":"Online advertising in recommendation platforms has gained significant attention, with a predominant focus on channel recommendation and budget allocation strategies. However, current offline reinforcement learning (RL) methods face substantial challenges when applied to sparse advertising scenarios, primarily due to severe overestimation, distributional shifts, and overlooking budget constraints. To address these issues, we propose MTORL, a novel multi-task offline RL model that targets two key objectives. First, we establish a Markov Decision Process (MDP) framework specific to the nuances of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.23090","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.23090/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.23090","created_at":"2026-07-05T11:34:10.150143+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.23090v2","created_at":"2026-07-05T11:34:10.150143+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.23090","created_at":"2026-07-05T11:34:10.150143+00:00"},{"alias_kind":"pith_short_12","alias_value":"3BHLKV43773L","created_at":"2026-07-05T11:34:10.150143+00:00"},{"alias_kind":"pith_short_16","alias_value":"3BHLKV43773L5FXO","created_at":"2026-07-05T11:34:10.150143+00:00"},{"alias_kind":"pith_short_8","alias_value":"3BHLKV43","created_at":"2026-07-05T11:34:10.150143+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB","json":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB.json","graph_json":"https://pith.science/api/pith-number/3BHLKV43773L5FXOD2SVJNU3BB/graph.json","events_json":"https://pith.science/api/pith-number/3BHLKV43773L5FXOD2SVJNU3BB/events.json","paper":"https://pith.science/paper/3BHLKV43"},"agent_actions":{"view_html":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB","download_json":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB.json","view_paper":"https://pith.science/paper/3BHLKV43","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.23090&json=true","fetch_graph":"https://pith.science/api/pith-number/3BHLKV43773L5FXOD2SVJNU3BB/graph.json","fetch_events":"https://pith.science/api/pith-number/3BHLKV43773L5FXOD2SVJNU3BB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB/action/storage_attestation","attest_author":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB/action/author_attestation","sign_citation":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB/action/citation_signature","submit_replication":"https://pith.science/pith/3BHLKV43773L5FXOD2SVJNU3BB/action/replication_record"}},"created_at":"2026-07-05T11:34:10.150143+00:00","updated_at":"2026-07-05T11:34:10.150143+00:00"}