{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:XG7ARARO5T3XHCWHZYBHDI7V6W","short_pith_number":"pith:XG7ARARO","schema_version":"1.0","canonical_sha256":"b9be08822eecf7738ac7ce0271a3f5f58f1539172987c5db6fda07da81a2605f","source":{"kind":"arxiv","id":"2607.18689","version":1},"attestation_state":"computed","paper":{"title":"Exposure-Based Reinforcement Learning to Rank","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.LG","authors_text":"Harrie Oosterhuis, Rolf Jagerman, Xuanhui Wang, Zhen Qin","submitted_at":"2026-07-21T04:11:20Z","abstract_excerpt":"Reinforcement learning (RL) methods for learning-to-rank (LTR) can optimize (almost) any ranking goal, e.g., from precision or discounted cumulative gain to fairness-of-exposure or ranking distillation. However, standard RL is ineffective and computationally costly due to the enormous action space in LTR settings. Existing methods reach computational efficiency through custom gradient computation algorithms, but they are very complex to implement and often clash with auto-differentiation. Consequently, existing RL for LTR is not attractive to many practitioners. We reconsider RL for LTR while "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.18689","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-21T04:11:20Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"4d96ecb9c80e99b5c9a3134f47e7bf3e7d496b5f7e72b21ccca6ce9650bc8d3e","abstract_canon_sha256":"f91526569d4e9c766008d0fe0822ce2fc3d1ab636a836b228e0d12bbf9979e12"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-22T01:23:06.903595Z","signature_b64":"BR5Lm8tA1NoR90y91CvFZG/PNV3RkH9kI7/4njmMHdKVy+3GP7DtATFOL3u51aqaWKjjBkV42OMmfGXAGCtWCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9be08822eecf7738ac7ce0271a3f5f58f1539172987c5db6fda07da81a2605f","last_reissued_at":"2026-07-22T01:23:06.902753Z","signature_status":"signed_v1","first_computed_at":"2026-07-22T01:23:06.902753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exposure-Based Reinforcement Learning to Rank","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.LG","authors_text":"Harrie Oosterhuis, Rolf Jagerman, Xuanhui Wang, Zhen Qin","submitted_at":"2026-07-21T04:11:20Z","abstract_excerpt":"Reinforcement learning (RL) methods for learning-to-rank (LTR) can optimize (almost) any ranking goal, e.g., from precision or discounted cumulative gain to fairness-of-exposure or ranking distillation. However, standard RL is ineffective and computationally costly due to the enormous action space in LTR settings. Existing methods reach computational efficiency through custom gradient computation algorithms, but they are very complex to implement and often clash with auto-differentiation. Consequently, existing RL for LTR is not attractive to many practitioners. We reconsider RL for LTR while "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.18689","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.18689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.18689","created_at":"2026-07-22T01:23:06.903184+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.18689v1","created_at":"2026-07-22T01:23:06.903184+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.18689","created_at":"2026-07-22T01:23:06.903184+00:00"},{"alias_kind":"pith_short_12","alias_value":"XG7ARARO5T3X","created_at":"2026-07-22T01:23:06.903184+00:00"},{"alias_kind":"pith_short_16","alias_value":"XG7ARARO5T3XHCWH","created_at":"2026-07-22T01:23:06.903184+00:00"},{"alias_kind":"pith_short_8","alias_value":"XG7ARARO","created_at":"2026-07-22T01:23:06.903184+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W","json":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W.json","graph_json":"https://pith.science/api/pith-number/XG7ARARO5T3XHCWHZYBHDI7V6W/graph.json","events_json":"https://pith.science/api/pith-number/XG7ARARO5T3XHCWHZYBHDI7V6W/events.json","paper":"https://pith.science/paper/XG7ARARO"},"agent_actions":{"view_html":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W","download_json":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W.json","view_paper":"https://pith.science/paper/XG7ARARO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.18689&json=true","fetch_graph":"https://pith.science/api/pith-number/XG7ARARO5T3XHCWHZYBHDI7V6W/graph.json","fetch_events":"https://pith.science/api/pith-number/XG7ARARO5T3XHCWHZYBHDI7V6W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W/action/storage_attestation","attest_author":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W/action/author_attestation","sign_citation":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W/action/citation_signature","submit_replication":"https://pith.science/pith/XG7ARARO5T3XHCWHZYBHDI7V6W/action/replication_record"}},"created_at":"2026-07-22T01:23:06.903184+00:00","updated_at":"2026-07-22T01:23:06.903184+00:00"}