{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q5ADV6G56VXQOG4QDJ4ECAS7JD","short_pith_number":"pith:Q5ADV6G5","schema_version":"1.0","canonical_sha256":"87403af8ddf56f071b901a7841025f48cf1734f34748c84045f8f2f0b7371543","source":{"kind":"arxiv","id":"2509.07253","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Information Retrieval Models on Complex Retrieval Tasks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.IR","authors_text":"Hamed Zamani, Julian Killingback","submitted_at":"2025-09-08T22:11:10Z","abstract_excerpt":"Large language models (LLMs) are incredible and versatile tools for text-based tasks that have enabled countless, previously unimaginable, applications. Retrieval models, in contrast, have not yet seen such capable general-purpose models emerge. To achieve this goal, retrieval models must be able to perform complex retrieval tasks, where queries contain multiple parts, constraints, or requirements in natural language. These tasks represent a natural progression from the simple, single-aspect queries that are used in the vast majority of existing, commonly used evaluation sets. Complex queries "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.07253","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.IR","submitted_at":"2025-09-08T22:11:10Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5d4c729d5b5c64300667aaaf1ff5059aad2f5be48f68f72c5e5be0809acfb6f4","abstract_canon_sha256":"f055731680dd236bc0e3edd9e1184c55ec0da46f86589626d4720362fc0dfa7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:18.451832Z","signature_b64":"y03M0jPvxuhcWtN9P8WLMi8enWkaQQVa4lwkFv2gqtj/z0PBKEQem2z6bOLzCrGIUIIP6QVmko0eHlheEQWNCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87403af8ddf56f071b901a7841025f48cf1734f34748c84045f8f2f0b7371543","last_reissued_at":"2026-07-05T12:07:18.451321Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:18.451321Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Information Retrieval Models on Complex Retrieval Tasks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.IR","authors_text":"Hamed Zamani, Julian Killingback","submitted_at":"2025-09-08T22:11:10Z","abstract_excerpt":"Large language models (LLMs) are incredible and versatile tools for text-based tasks that have enabled countless, previously unimaginable, applications. Retrieval models, in contrast, have not yet seen such capable general-purpose models emerge. To achieve this goal, retrieval models must be able to perform complex retrieval tasks, where queries contain multiple parts, constraints, or requirements in natural language. These tasks represent a natural progression from the simple, single-aspect queries that are used in the vast majority of existing, commonly used evaluation sets. Complex queries "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.07253","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.07253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.07253","created_at":"2026-07-05T12:07:18.451385+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.07253v1","created_at":"2026-07-05T12:07:18.451385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.07253","created_at":"2026-07-05T12:07:18.451385+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q5ADV6G56VXQ","created_at":"2026-07-05T12:07:18.451385+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q5ADV6G56VXQOG4Q","created_at":"2026-07-05T12:07:18.451385+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q5ADV6G5","created_at":"2026-07-05T12:07:18.451385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13137","citing_title":"LeanSearch v2: Global Premise Retrieval for Lean 4 Theorem Proving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13137","citing_title":"LeanSearch v2: Global Premise Retrieval for Lean 4 Theorem Proving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27577","citing_title":"Reproducing Adaptive Reranking for Reasoning-Intensive IR","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22661","citing_title":"Can QPP Choose the Right Query Variant? Evaluating Query Variant Selection for RAG Pipelines","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21096","citing_title":"Multilingual and Domain-Agnostic Tip-of-the-Tongue Query Generation for Simulated Evaluation","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD","json":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD.json","graph_json":"https://pith.science/api/pith-number/Q5ADV6G56VXQOG4QDJ4ECAS7JD/graph.json","events_json":"https://pith.science/api/pith-number/Q5ADV6G56VXQOG4QDJ4ECAS7JD/events.json","paper":"https://pith.science/paper/Q5ADV6G5"},"agent_actions":{"view_html":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD","download_json":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD.json","view_paper":"https://pith.science/paper/Q5ADV6G5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.07253&json=true","fetch_graph":"https://pith.science/api/pith-number/Q5ADV6G56VXQOG4QDJ4ECAS7JD/graph.json","fetch_events":"https://pith.science/api/pith-number/Q5ADV6G56VXQOG4QDJ4ECAS7JD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD/action/storage_attestation","attest_author":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD/action/author_attestation","sign_citation":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD/action/citation_signature","submit_replication":"https://pith.science/pith/Q5ADV6G56VXQOG4QDJ4ECAS7JD/action/replication_record"}},"created_at":"2026-07-05T12:07:18.451385+00:00","updated_at":"2026-07-05T12:07:18.451385+00:00"}