{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YP3NH5NEPLZ5IHOCUMNDJKGO3M","short_pith_number":"pith:YP3NH5NE","schema_version":"1.0","canonical_sha256":"c3f6d3f5a47af3d41dc2a31a34a8cedb1fd56a2a08d8b5c1a08c553e446db31f","source":{"kind":"arxiv","id":"2310.04407","version":2},"attestation_state":"computed","paper":{"title":"Policy-Gradient Training of Language Models for Ranking","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Ge Gao, Jonathan D. Chang, Kiant\\'e Brantley, Thorsten Joachim","submitted_at":"2023-10-06T17:55:23Z","abstract_excerpt":"Text retrieval plays a crucial role in incorporating factual knowledge for decision making into language processing pipelines, ranging from chat-based web search to question answering systems. Current state-of-the-art text retrieval models leverage pre-trained large language models (LLMs) to achieve competitive performance, but training LLM-based retrievers via typical contrastive losses requires intricate heuristics, including selecting hard negatives and using additional supervision as learning signals. This reliance on heuristics stems from the fact that the contrastive loss itself is heuri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.04407","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-06T17:55:23Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG"],"title_canon_sha256":"cb9b7df21f72e8f440b3d8ab5b6250d41994e737bee3447b67b79811fe447662","abstract_canon_sha256":"deb7309dae114e3e3cb35414e9e8eb6da4d1702501fe1e1cec05db3c28653e08"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:19.701862Z","signature_b64":"mK6dusYx6qUvl5i4qQs1s8km5PRo4FpoY1u0z1WGdO1VTFDNh2Gnyl+Xhy3oqEISSMY5ApHa9wJG0KyvKrffDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3f6d3f5a47af3d41dc2a31a34a8cedb1fd56a2a08d8b5c1a08c553e446db31f","last_reissued_at":"2026-07-05T09:39:19.701310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:19.701310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Policy-Gradient Training of Language Models for Ranking","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Ge Gao, Jonathan D. Chang, Kiant\\'e Brantley, Thorsten Joachim","submitted_at":"2023-10-06T17:55:23Z","abstract_excerpt":"Text retrieval plays a crucial role in incorporating factual knowledge for decision making into language processing pipelines, ranging from chat-based web search to question answering systems. Current state-of-the-art text retrieval models leverage pre-trained large language models (LLMs) to achieve competitive performance, but training LLM-based retrievers via typical contrastive losses requires intricate heuristics, including selecting hard negatives and using additional supervision as learning signals. This reliance on heuristics stems from the fact that the contrastive loss itself is heuri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.04407","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.04407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.04407","created_at":"2026-07-05T09:39:19.701373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.04407v2","created_at":"2026-07-05T09:39:19.701373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.04407","created_at":"2026-07-05T09:39:19.701373+00:00"},{"alias_kind":"pith_short_12","alias_value":"YP3NH5NEPLZ5","created_at":"2026-07-05T09:39:19.701373+00:00"},{"alias_kind":"pith_short_16","alias_value":"YP3NH5NEPLZ5IHOC","created_at":"2026-07-05T09:39:19.701373+00:00"},{"alias_kind":"pith_short_8","alias_value":"YP3NH5NE","created_at":"2026-07-05T09:39:19.701373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26385","citing_title":"Credit-assigned Policy Gradient for Early Stage Retrieval in Two-stage Ranking","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12995","citing_title":"F-GRPO: Factorized Group-Relative Policy Optimization for Unified Candidate Generation and Ranking","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M","json":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M.json","graph_json":"https://pith.science/api/pith-number/YP3NH5NEPLZ5IHOCUMNDJKGO3M/graph.json","events_json":"https://pith.science/api/pith-number/YP3NH5NEPLZ5IHOCUMNDJKGO3M/events.json","paper":"https://pith.science/paper/YP3NH5NE"},"agent_actions":{"view_html":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M","download_json":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M.json","view_paper":"https://pith.science/paper/YP3NH5NE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.04407&json=true","fetch_graph":"https://pith.science/api/pith-number/YP3NH5NEPLZ5IHOCUMNDJKGO3M/graph.json","fetch_events":"https://pith.science/api/pith-number/YP3NH5NEPLZ5IHOCUMNDJKGO3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M/action/storage_attestation","attest_author":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M/action/author_attestation","sign_citation":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M/action/citation_signature","submit_replication":"https://pith.science/pith/YP3NH5NEPLZ5IHOCUMNDJKGO3M/action/replication_record"}},"created_at":"2026-07-05T09:39:19.701373+00:00","updated_at":"2026-07-05T09:39:19.701373+00:00"}