{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AJ6WABBVEMPYWARF3N73DL37KH","short_pith_number":"pith:AJ6WABBV","schema_version":"1.0","canonical_sha256":"027d600435231f8b0225db7fb1af7f51c6e13af0c639d99e6488e11ee01ee4df","source":{"kind":"arxiv","id":"2307.04571","version":1},"attestation_state":"computed","paper":{"title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Biao Li, Chongming Gao, Jiawei Chen, Kexin Huang, Peng Jiang, Shiqi Wang, Xiangnan He, Yuan Zhang, Zhong Zhang","submitted_at":"2023-07-10T14:03:34Z","abstract_excerpt":"Offline reinforcement learning (RL), a technology that offline learns a policy from logged data without the need to interact with online environments, has become a favorable choice in decision-making processes like interactive recommendation. Offline RL faces the value overestimation problem. To address it, existing methods employ conservatism, e.g., by constraining the learned policy to be close to behavior policies or punishing the rarely visited state-action pairs. However, when applying such offline RL to recommendation, it will cause a severe Matthew effect, i.e., the rich get richer and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.04571","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2023-07-10T14:03:34Z","cross_cats_sorted":[],"title_canon_sha256":"0ba41332c911f3924cffa01c47746a9250c66fa8753595085b9babb752462994","abstract_canon_sha256":"5608f695506678b1151b98b90f5e68315387bba86a385358658604b25059c319"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:29:15.185098Z","signature_b64":"ViILuqJMRRWEKzZnhV//1jy4LYSQYWAODDCodL1WbeYgpuSXAbJZphFFoVtiN5qNJNHfhU+9ITN9RZHkWhvpAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"027d600435231f8b0225db7fb1af7f51c6e13af0c639d99e6488e11ee01ee4df","last_reissued_at":"2026-07-05T06:29:15.184674Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:29:15.184674Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Biao Li, Chongming Gao, Jiawei Chen, Kexin Huang, Peng Jiang, Shiqi Wang, Xiangnan He, Yuan Zhang, Zhong Zhang","submitted_at":"2023-07-10T14:03:34Z","abstract_excerpt":"Offline reinforcement learning (RL), a technology that offline learns a policy from logged data without the need to interact with online environments, has become a favorable choice in decision-making processes like interactive recommendation. Offline RL faces the value overestimation problem. To address it, existing methods employ conservatism, e.g., by constraining the learned policy to be close to behavior policies or punishing the rarely visited state-action pairs. However, when applying such offline RL to recommendation, it will cause a severe Matthew effect, i.e., the rich get richer and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.04571","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.04571/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.04571","created_at":"2026-07-05T06:29:15.184741+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.04571v1","created_at":"2026-07-05T06:29:15.184741+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.04571","created_at":"2026-07-05T06:29:15.184741+00:00"},{"alias_kind":"pith_short_12","alias_value":"AJ6WABBVEMPY","created_at":"2026-07-05T06:29:15.184741+00:00"},{"alias_kind":"pith_short_16","alias_value":"AJ6WABBVEMPYWARF","created_at":"2026-07-05T06:29:15.184741+00:00"},{"alias_kind":"pith_short_8","alias_value":"AJ6WABBV","created_at":"2026-07-05T06:29:15.184741+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH","json":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH.json","graph_json":"https://pith.science/api/pith-number/AJ6WABBVEMPYWARF3N73DL37KH/graph.json","events_json":"https://pith.science/api/pith-number/AJ6WABBVEMPYWARF3N73DL37KH/events.json","paper":"https://pith.science/paper/AJ6WABBV"},"agent_actions":{"view_html":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH","download_json":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH.json","view_paper":"https://pith.science/paper/AJ6WABBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.04571&json=true","fetch_graph":"https://pith.science/api/pith-number/AJ6WABBVEMPYWARF3N73DL37KH/graph.json","fetch_events":"https://pith.science/api/pith-number/AJ6WABBVEMPYWARF3N73DL37KH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH/action/storage_attestation","attest_author":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH/action/author_attestation","sign_citation":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH/action/citation_signature","submit_replication":"https://pith.science/pith/AJ6WABBVEMPYWARF3N73DL37KH/action/replication_record"}},"created_at":"2026-07-05T06:29:15.184741+00:00","updated_at":"2026-07-05T06:29:15.184741+00:00"}