{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:KIK5NYZL7QCTBOHVJ54BOB3HED","short_pith_number":"pith:KIK5NYZL","schema_version":"1.0","canonical_sha256":"5215d6e32bfc0530b8f54f7817076720c28ae0654481457b6adb57f49521509c","source":{"kind":"arxiv","id":"2109.12509","version":4},"attestation_state":"computed","paper":{"title":"Deep Exploration for Recommendation Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.IR","authors_text":"Benjamin Van Roy, Zheqing Zhu","submitted_at":"2021-09-26T06:54:26Z","abstract_excerpt":"Modern recommendation systems ought to benefit by probing for and learning from delayed feedback. Research has tended to focus on learning from a user's response to a single recommendation. Such work, which leverages methods of supervised and bandit learning, forgoes learning from the user's subsequent behavior. Where past work has aimed to learn from subsequent behavior, there has been a lack of effective methods for probing to elicit informative delayed feedback. Effective exploration through probing for delayed feedback becomes particularly challenging when rewards are sparse. To address th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.12509","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2021-09-26T06:54:26Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"bb948ad4a5ad5e5c175a4766610eb71e8261a300d351029d7bf78ecc827dd847","abstract_canon_sha256":"2eadb125aa5f7975c20c9ef5fdd7bdc4b1aada443addcd8e6de488f2d3bbf6d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:42.272776Z","signature_b64":"yN7N0DsuHq7KXkwQ1UbK3009NuzO3JBQ5IK3VTBntig9kP7MVuZ3/pSfC0DX9cnpTkkx4CoJh8UigJ2kuKUBBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5215d6e32bfc0530b8f54f7817076720c28ae0654481457b6adb57f49521509c","last_reissued_at":"2026-07-05T06:35:42.272240Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:42.272240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Exploration for Recommendation Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.IR","authors_text":"Benjamin Van Roy, Zheqing Zhu","submitted_at":"2021-09-26T06:54:26Z","abstract_excerpt":"Modern recommendation systems ought to benefit by probing for and learning from delayed feedback. Research has tended to focus on learning from a user's response to a single recommendation. Such work, which leverages methods of supervised and bandit learning, forgoes learning from the user's subsequent behavior. Where past work has aimed to learn from subsequent behavior, there has been a lack of effective methods for probing to elicit informative delayed feedback. Effective exploration through probing for delayed feedback becomes particularly challenging when rewards are sparse. To address th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.12509","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.12509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.12509","created_at":"2026-07-05T06:35:42.272297+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.12509v4","created_at":"2026-07-05T06:35:42.272297+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.12509","created_at":"2026-07-05T06:35:42.272297+00:00"},{"alias_kind":"pith_short_12","alias_value":"KIK5NYZL7QCT","created_at":"2026-07-05T06:35:42.272297+00:00"},{"alias_kind":"pith_short_16","alias_value":"KIK5NYZL7QCTBOHV","created_at":"2026-07-05T06:35:42.272297+00:00"},{"alias_kind":"pith_short_8","alias_value":"KIK5NYZL","created_at":"2026-07-05T06:35:42.272297+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED","json":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED.json","graph_json":"https://pith.science/api/pith-number/KIK5NYZL7QCTBOHVJ54BOB3HED/graph.json","events_json":"https://pith.science/api/pith-number/KIK5NYZL7QCTBOHVJ54BOB3HED/events.json","paper":"https://pith.science/paper/KIK5NYZL"},"agent_actions":{"view_html":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED","download_json":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED.json","view_paper":"https://pith.science/paper/KIK5NYZL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.12509&json=true","fetch_graph":"https://pith.science/api/pith-number/KIK5NYZL7QCTBOHVJ54BOB3HED/graph.json","fetch_events":"https://pith.science/api/pith-number/KIK5NYZL7QCTBOHVJ54BOB3HED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED/action/storage_attestation","attest_author":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED/action/author_attestation","sign_citation":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED/action/citation_signature","submit_replication":"https://pith.science/pith/KIK5NYZL7QCTBOHVJ54BOB3HED/action/replication_record"}},"created_at":"2026-07-05T06:35:42.272297+00:00","updated_at":"2026-07-05T06:35:42.272297+00:00"}