{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DZSG3BXMUYK75HYE3FO3HJJLOE","short_pith_number":"pith:DZSG3BXM","schema_version":"1.0","canonical_sha256":"1e646d86eca615fe9f04d95db3a52b7112859696be4b9b4af9be4ae8b2606824","source":{"kind":"arxiv","id":"2403.15246","version":3},"attestation_state":"computed","paper":{"title":"FollowIR: Evaluating and Teaching Information Retrieval Models to Follow Instructions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.IR","authors_text":"Arman Cohan, Benjamin Chang, Benjamin Van Durme, Dawn Lawrie, Kyle Lo, Luca Soldaini, Orion Weller, Sean MacAvaney","submitted_at":"2024-03-22T14:42:29Z","abstract_excerpt":"Modern Language Models (LMs) are capable of following long and complex instructions that enable a large and diverse set of user requests. While Information Retrieval (IR) models use these LMs as the backbone of their architectures, virtually none of them allow users to provide detailed instructions alongside queries, thus limiting their ability to satisfy complex information needs. In this work, we study the use of instructions in IR systems. First, we introduce our dataset FollowIR, which contains a rigorous instruction evaluation benchmark as well as a training set for helping IR models lear"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.15246","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-03-22T14:42:29Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"540e2dbf28636ed661bed493c302cfc9f5e131459e544867abebf90f0d1a50e1","abstract_canon_sha256":"e191a9eaec09a6af34b9773efe2c8037b280a670ec056c3d7616ceeefdb18f4e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:16:26.170658Z","signature_b64":"FavxKXBUKp0URp4QrqU1renBB3ufLsaXBwtortlHHinhfrJRBFm8UsJQDhLIqOEfUVzkOWIdny5rfe9Pt2FPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e646d86eca615fe9f04d95db3a52b7112859696be4b9b4af9be4ae8b2606824","last_reissued_at":"2026-07-05T08:16:26.170178Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:16:26.170178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FollowIR: Evaluating and Teaching Information Retrieval Models to Follow Instructions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.IR","authors_text":"Arman Cohan, Benjamin Chang, Benjamin Van Durme, Dawn Lawrie, Kyle Lo, Luca Soldaini, Orion Weller, Sean MacAvaney","submitted_at":"2024-03-22T14:42:29Z","abstract_excerpt":"Modern Language Models (LMs) are capable of following long and complex instructions that enable a large and diverse set of user requests. While Information Retrieval (IR) models use these LMs as the backbone of their architectures, virtually none of them allow users to provide detailed instructions alongside queries, thus limiting their ability to satisfy complex information needs. In this work, we study the use of instructions in IR systems. First, we introduce our dataset FollowIR, which contains a rigorous instruction evaluation benchmark as well as a training set for helping IR models lear"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.15246","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.15246/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.15246","created_at":"2026-07-05T08:16:26.170233+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.15246v3","created_at":"2026-07-05T08:16:26.170233+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.15246","created_at":"2026-07-05T08:16:26.170233+00:00"},{"alias_kind":"pith_short_12","alias_value":"DZSG3BXMUYK7","created_at":"2026-07-05T08:16:26.170233+00:00"},{"alias_kind":"pith_short_16","alias_value":"DZSG3BXMUYK75HYE","created_at":"2026-07-05T08:16:26.170233+00:00"},{"alias_kind":"pith_short_8","alias_value":"DZSG3BXM","created_at":"2026-07-05T08:16:26.170233+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22778","citing_title":"HAKARI-Bench: A Lightweight Benchmark for Comparing Retrieval Architectures and Efficiency Settings under Unified Conditions","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27037","citing_title":"Hypencoder Revisited: Reproducibility and Analysis of Non-Linear Scoring for First-Stage Retrieval","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05176","citing_title":"Qwen3 Embedding: Advancing Text Embedding and Reranking Through Foundation Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE","json":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE.json","graph_json":"https://pith.science/api/pith-number/DZSG3BXMUYK75HYE3FO3HJJLOE/graph.json","events_json":"https://pith.science/api/pith-number/DZSG3BXMUYK75HYE3FO3HJJLOE/events.json","paper":"https://pith.science/paper/DZSG3BXM"},"agent_actions":{"view_html":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE","download_json":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE.json","view_paper":"https://pith.science/paper/DZSG3BXM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.15246&json=true","fetch_graph":"https://pith.science/api/pith-number/DZSG3BXMUYK75HYE3FO3HJJLOE/graph.json","fetch_events":"https://pith.science/api/pith-number/DZSG3BXMUYK75HYE3FO3HJJLOE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE/action/storage_attestation","attest_author":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE/action/author_attestation","sign_citation":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE/action/citation_signature","submit_replication":"https://pith.science/pith/DZSG3BXMUYK75HYE3FO3HJJLOE/action/replication_record"}},"created_at":"2026-07-05T08:16:26.170233+00:00","updated_at":"2026-07-05T08:16:26.170233+00:00"}