{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LTFODNVIPCEDGO5YIT4SZMG5XD","short_pith_number":"pith:LTFODNVI","schema_version":"1.0","canonical_sha256":"5ccae1b6a87888333bb844f92cb0ddb8c51ede8f4c1d3f563defcddaa8df94c3","source":{"kind":"arxiv","id":"2503.14887","version":2},"attestation_state":"computed","paper":{"title":"Pseudo Relevance Feedback is Enough to Close the Gap Between Small and Large Dense Retrieval Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Bevan Koopman, Guido Zuccon, Hang Li, Xiao Wang","submitted_at":"2025-03-19T04:30:20Z","abstract_excerpt":"Scaling dense retrievers to larger large language model (LLM) backbones has been a dominant strategy for improving their retrieval effectiveness. However, this has substantial cost implications: larger backbones require more expensive hardware (e.g. GPUs with more memory) and lead to higher indexing and querying costs (latency, energy consumption). In this paper, we challenge this paradigm by introducing PromptPRF, a feature-based pseudo-relevance feedback (PRF) framework that enables small LLM-based dense retrievers to achieve effectiveness comparable to much larger models.\n  PromptPRF uses L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.14887","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2025-03-19T04:30:20Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e6cddd1795ac1f28ed8c550e010d69f1b89d6e90c7bd8389f451d0176adb4223","abstract_canon_sha256":"32f3a6dcc48caeff982b8727d62af912dc4f79cb6599595b5219744e5fe2a69f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:56.715840Z","signature_b64":"k7zZxFH67B0k9o1uChZ1f565DFi/1L8TcNnWeTAK8vWE2sK61ip074rbEzyM/CzucFVhvg4ptSjb6VJOTC1CBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ccae1b6a87888333bb844f92cb0ddb8c51ede8f4c1d3f563defcddaa8df94c3","last_reissued_at":"2026-07-05T11:16:56.715275Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:56.715275Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pseudo Relevance Feedback is Enough to Close the Gap Between Small and Large Dense Retrieval Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Bevan Koopman, Guido Zuccon, Hang Li, Xiao Wang","submitted_at":"2025-03-19T04:30:20Z","abstract_excerpt":"Scaling dense retrievers to larger large language model (LLM) backbones has been a dominant strategy for improving their retrieval effectiveness. However, this has substantial cost implications: larger backbones require more expensive hardware (e.g. GPUs with more memory) and lead to higher indexing and querying costs (latency, energy consumption). In this paper, we challenge this paradigm by introducing PromptPRF, a feature-based pseudo-relevance feedback (PRF) framework that enables small LLM-based dense retrievers to achieve effectiveness comparable to much larger models.\n  PromptPRF uses L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.14887","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.14887/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.14887","created_at":"2026-07-05T11:16:56.715346+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.14887v2","created_at":"2026-07-05T11:16:56.715346+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.14887","created_at":"2026-07-05T11:16:56.715346+00:00"},{"alias_kind":"pith_short_12","alias_value":"LTFODNVIPCED","created_at":"2026-07-05T11:16:56.715346+00:00"},{"alias_kind":"pith_short_16","alias_value":"LTFODNVIPCEDGO5Y","created_at":"2026-07-05T11:16:56.715346+00:00"},{"alias_kind":"pith_short_8","alias_value":"LTFODNVI","created_at":"2026-07-05T11:16:56.715346+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01070","citing_title":"Test-Time Training for Zero-Resource Dense Retrieval Reranking","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11374","citing_title":"Test-Time Compute for Frozen Embedding Models through Agentic Program Search","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07794","citing_title":"Query Expansion in the Age of Pre-trained and Large Language Models: A Comprehensive Survey","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11374","citing_title":"Test-Time Compute for Frozen Embedding Models through Agentic Program Search","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11374","citing_title":"Test-Time Compute for Frozen Embedding Models through Agentic Program Search","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD","json":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD.json","graph_json":"https://pith.science/api/pith-number/LTFODNVIPCEDGO5YIT4SZMG5XD/graph.json","events_json":"https://pith.science/api/pith-number/LTFODNVIPCEDGO5YIT4SZMG5XD/events.json","paper":"https://pith.science/paper/LTFODNVI"},"agent_actions":{"view_html":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD","download_json":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD.json","view_paper":"https://pith.science/paper/LTFODNVI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.14887&json=true","fetch_graph":"https://pith.science/api/pith-number/LTFODNVIPCEDGO5YIT4SZMG5XD/graph.json","fetch_events":"https://pith.science/api/pith-number/LTFODNVIPCEDGO5YIT4SZMG5XD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD/action/storage_attestation","attest_author":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD/action/author_attestation","sign_citation":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD/action/citation_signature","submit_replication":"https://pith.science/pith/LTFODNVIPCEDGO5YIT4SZMG5XD/action/replication_record"}},"created_at":"2026-07-05T11:16:56.715346+00:00","updated_at":"2026-07-05T11:16:56.715346+00:00"}