{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B6FEIIQMIN6PXLIZOF4NZSKZOQ","short_pith_number":"pith:B6FEIIQM","schema_version":"1.0","canonical_sha256":"0f8a44220c437cfbad197178dcc959741922c330131289e53e4f1eb46699e26d","source":{"kind":"arxiv","id":"2504.20595","version":1},"attestation_state":"computed","paper":{"title":"ReasonIR: Training Retrievers for Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.AI","authors_text":"Bryan Kian Hsiang Low, Daniela Rus, Luke Zettlemoyer, Niklas Muennighoff, Pang Wei Koh, Rui Qiao, Rulin Shao, Sewon Min, Varsha Kishore, Wen-tau Yih, Xi Victoria Lin","submitted_at":"2025-04-29T09:49:28Z","abstract_excerpt":"We present ReasonIR-8B, the first retriever specifically trained for general reasoning tasks. Existing retrievers have shown limited gains on reasoning tasks, in part because existing training datasets focus on short factual queries tied to documents that straightforwardly answer them. We develop a synthetic data generation pipeline that, for each document, our pipeline creates a challenging and relevant query, along with a plausibly related but ultimately unhelpful hard negative. By training on a mixture of our synthetic data and existing public data, ReasonIR-8B achieves a new state-of-the-a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20595","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-04-29T09:49:28Z","cross_cats_sorted":["cs.CL","cs.IR","cs.LG"],"title_canon_sha256":"93aef51d60b9bf27d6e014326ad92b7328e06bc2cdc3437855e100f31ac2cdf1","abstract_canon_sha256":"4e42567ef91990d2a565832742acfda5703599dbd03c3d477bcff38b938eb905"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:45.312602Z","signature_b64":"qNcLMjk7HRIfHoPpqnkgUTIP+5T1/lK2WeP2g2WI/cQ/ZOs4xzopvpBhFECJEHIojlvB/Rnf68ymywHSwG4kAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f8a44220c437cfbad197178dcc959741922c330131289e53e4f1eb46699e26d","last_reissued_at":"2026-07-05T10:55:45.312026Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:45.312026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReasonIR: Training Retrievers for Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.AI","authors_text":"Bryan Kian Hsiang Low, Daniela Rus, Luke Zettlemoyer, Niklas Muennighoff, Pang Wei Koh, Rui Qiao, Rulin Shao, Sewon Min, Varsha Kishore, Wen-tau Yih, Xi Victoria Lin","submitted_at":"2025-04-29T09:49:28Z","abstract_excerpt":"We present ReasonIR-8B, the first retriever specifically trained for general reasoning tasks. Existing retrievers have shown limited gains on reasoning tasks, in part because existing training datasets focus on short factual queries tied to documents that straightforwardly answer them. We develop a synthetic data generation pipeline that, for each document, our pipeline creates a challenging and relevant query, along with a plausibly related but ultimately unhelpful hard negative. By training on a mixture of our synthetic data and existing public data, ReasonIR-8B achieves a new state-of-the-a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20595","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20595","created_at":"2026-07-05T10:55:45.312091+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20595v1","created_at":"2026-07-05T10:55:45.312091+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20595","created_at":"2026-07-05T10:55:45.312091+00:00"},{"alias_kind":"pith_short_12","alias_value":"B6FEIIQMIN6P","created_at":"2026-07-05T10:55:45.312091+00:00"},{"alias_kind":"pith_short_16","alias_value":"B6FEIIQMIN6PXLIZ","created_at":"2026-07-05T10:55:45.312091+00:00"},{"alias_kind":"pith_short_8","alias_value":"B6FEIIQM","created_at":"2026-07-05T10:55:45.312091+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13137","citing_title":"LeanSearch v2: Global Premise Retrieval for Lean 4 Theorem Proving","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13137","citing_title":"LeanSearch v2: Global Premise Retrieval for Lean 4 Theorem Proving","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13521","citing_title":"Granite Embedding Multilingual R2 Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01348","citing_title":"Procedural Knowledge at Scale Improves Reasoning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03676","citing_title":"Are LLM-Based Retrievers Worth Their Cost? An Empirical Study of Efficiency, Robustness, and Reasoning Overhead","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12370","citing_title":"Context Convergence Improves Answering Inferential Questions","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00400","citing_title":"FollowTable: A Benchmark for Instruction-Following Table Retrieval","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00063","citing_title":"A Survey of Reasoning-Intensive Retrieval: Progress and Challenges","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07220","citing_title":"HIVE: Query, Hypothesize, Verify An LLM Framework for Multimodal Reasoning-Intensive Retrieval","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07201","citing_title":"BRIDGE: Multimodal-to-Text Retrieval via Reinforcement-Learned Query Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07079","citing_title":"MARVEL: Multimodal Adaptive Reasoning-intensiVe Expand-rerank and retrievaL","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20144","citing_title":"An Agentic Approach to Metadata Reasoning","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ","json":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ.json","graph_json":"https://pith.science/api/pith-number/B6FEIIQMIN6PXLIZOF4NZSKZOQ/graph.json","events_json":"https://pith.science/api/pith-number/B6FEIIQMIN6PXLIZOF4NZSKZOQ/events.json","paper":"https://pith.science/paper/B6FEIIQM"},"agent_actions":{"view_html":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ","download_json":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ.json","view_paper":"https://pith.science/paper/B6FEIIQM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20595&json=true","fetch_graph":"https://pith.science/api/pith-number/B6FEIIQMIN6PXLIZOF4NZSKZOQ/graph.json","fetch_events":"https://pith.science/api/pith-number/B6FEIIQMIN6PXLIZOF4NZSKZOQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ/action/storage_attestation","attest_author":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ/action/author_attestation","sign_citation":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ/action/citation_signature","submit_replication":"https://pith.science/pith/B6FEIIQMIN6PXLIZOF4NZSKZOQ/action/replication_record"}},"created_at":"2026-07-05T10:55:45.312091+00:00","updated_at":"2026-07-05T10:55:45.312091+00:00"}