{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5UTC4TNSKEGCMDRFW3B4AHABWH","short_pith_number":"pith:5UTC4TNS","schema_version":"1.0","canonical_sha256":"ed262e4db2510c260e25b6c3c01c01b1d405f10d90a706ac00475399e680cf4b","source":{"kind":"arxiv","id":"2212.11431","version":2},"attestation_state":"computed","paper":{"title":"Local Policy Improvement for Recommender Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.LG","authors_text":"Dawen Liang, Nikos Vlassis","submitted_at":"2022-12-22T00:47:40Z","abstract_excerpt":"Recommender systems predict what items a user will interact with next, based on their past interactions. The problem is often approached through supervised learning, but recent advancements have shifted towards policy optimization of rewards (e.g., user engagement). One challenge with the latter is policy mismatch: we are only able to train a new policy given data collected from a previously-deployed policy. The conventional way to address this problem is through importance sampling correction, but this comes with practical limitations. We suggest an alternative approach of local policy improv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.11431","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-22T00:47:40Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"8ca73747f12b6f495de89ece7524649df4b3fc9b1e5d9d330e677459273c6bcb","abstract_canon_sha256":"ab10578e05e54554c8f6e4921ed4543a7d7d74466245a16de742675a0542b08e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:04:52.999979Z","signature_b64":"6ua5Pv+hJDz8TW23+812FrmooEsnPHVN6IPRONmOLA8ISltVpAYdN3pmB9fsJvh2WIjgt8uSIP+jlO3DQlrSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed262e4db2510c260e25b6c3c01c01b1d405f10d90a706ac00475399e680cf4b","last_reissued_at":"2026-07-05T06:04:52.999641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:04:52.999641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Local Policy Improvement for Recommender Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.LG","authors_text":"Dawen Liang, Nikos Vlassis","submitted_at":"2022-12-22T00:47:40Z","abstract_excerpt":"Recommender systems predict what items a user will interact with next, based on their past interactions. The problem is often approached through supervised learning, but recent advancements have shifted towards policy optimization of rewards (e.g., user engagement). One challenge with the latter is policy mismatch: we are only able to train a new policy given data collected from a previously-deployed policy. The conventional way to address this problem is through importance sampling correction, but this comes with practical limitations. We suggest an alternative approach of local policy improv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.11431","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.11431/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.11431","created_at":"2026-07-05T06:04:52.999701+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.11431v2","created_at":"2026-07-05T06:04:52.999701+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.11431","created_at":"2026-07-05T06:04:52.999701+00:00"},{"alias_kind":"pith_short_12","alias_value":"5UTC4TNSKEGC","created_at":"2026-07-05T06:04:52.999701+00:00"},{"alias_kind":"pith_short_16","alias_value":"5UTC4TNSKEGCMDRF","created_at":"2026-07-05T06:04:52.999701+00:00"},{"alias_kind":"pith_short_8","alias_value":"5UTC4TNS","created_at":"2026-07-05T06:04:52.999701+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.00816","citing_title":"Exponential Reward Weighting for Fine-Tuning Generative Recommenders under Sparse and Noisy Feedback","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH","json":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH.json","graph_json":"https://pith.science/api/pith-number/5UTC4TNSKEGCMDRFW3B4AHABWH/graph.json","events_json":"https://pith.science/api/pith-number/5UTC4TNSKEGCMDRFW3B4AHABWH/events.json","paper":"https://pith.science/paper/5UTC4TNS"},"agent_actions":{"view_html":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH","download_json":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH.json","view_paper":"https://pith.science/paper/5UTC4TNS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.11431&json=true","fetch_graph":"https://pith.science/api/pith-number/5UTC4TNSKEGCMDRFW3B4AHABWH/graph.json","fetch_events":"https://pith.science/api/pith-number/5UTC4TNSKEGCMDRFW3B4AHABWH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH/action/storage_attestation","attest_author":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH/action/author_attestation","sign_citation":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH/action/citation_signature","submit_replication":"https://pith.science/pith/5UTC4TNSKEGCMDRFW3B4AHABWH/action/replication_record"}},"created_at":"2026-07-05T06:04:52.999701+00:00","updated_at":"2026-07-05T06:04:52.999701+00:00"}