{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JDOQK6OPSZI4BGT3T5LMU46NTL","short_pith_number":"pith:JDOQK6OP","schema_version":"1.0","canonical_sha256":"48dd0579cf9651c09a7b9f56ca73cd9ad5b16ea78c5a86543dbbf4abe5263996","source":{"kind":"arxiv","id":"2501.11628","version":1},"attestation_state":"computed","paper":{"title":"Investigating the Scalability of Approximate Sparse Retrieval Algorithms to Massive Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Cosimo Rulli, Franco Maria Nardini, Leonardo Venuta, Rossano Venturini, Sebastian Bruch","submitted_at":"2025-01-20T17:59:21Z","abstract_excerpt":"Learned sparse text embeddings have gained popularity due to their effectiveness in top-k retrieval and inherent interpretability. Their distributional idiosyncrasies, however, have long hindered their use in real-world retrieval systems. That changed with the recent development of approximate algorithms that leverage the distributional properties of sparse embeddings to speed up retrieval. Nonetheless, in much of the existing literature, evaluation has been limited to datasets with only a few million documents such as MSMARCO. It remains unclear how these systems behave on much larger dataset"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11628","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2025-01-20T17:59:21Z","cross_cats_sorted":[],"title_canon_sha256":"0ff9aa6e7717c1f0d275c540cae1460443f82c5bd9b41744b0fc0c568b2e578d","abstract_canon_sha256":"9da21414ba9707c61c4a15a54d3226b1f7988095d057dae3b5971b579358df97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:03:11.734647Z","signature_b64":"1LfMQxOJK3/+6rvdXEoCbAHo6YvJ02OmQ8GbqOqsNDOMl8hM5BOMqEiHgkveFsDQY0b6O6pvEv+IHG13TaneAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48dd0579cf9651c09a7b9f56ca73cd9ad5b16ea78c5a86543dbbf4abe5263996","last_reissued_at":"2026-07-05T10:03:11.734266Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:03:11.734266Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating the Scalability of Approximate Sparse Retrieval Algorithms to Massive Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Cosimo Rulli, Franco Maria Nardini, Leonardo Venuta, Rossano Venturini, Sebastian Bruch","submitted_at":"2025-01-20T17:59:21Z","abstract_excerpt":"Learned sparse text embeddings have gained popularity due to their effectiveness in top-k retrieval and inherent interpretability. Their distributional idiosyncrasies, however, have long hindered their use in real-world retrieval systems. That changed with the recent development of approximate algorithms that leverage the distributional properties of sparse embeddings to speed up retrieval. Nonetheless, in much of the existing literature, evaluation has been limited to datasets with only a few million documents such as MSMARCO. It remains unclear how these systems behave on much larger dataset"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11628","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11628","created_at":"2026-07-05T10:03:11.734322+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11628v1","created_at":"2026-07-05T10:03:11.734322+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11628","created_at":"2026-07-05T10:03:11.734322+00:00"},{"alias_kind":"pith_short_12","alias_value":"JDOQK6OPSZI4","created_at":"2026-07-05T10:03:11.734322+00:00"},{"alias_kind":"pith_short_16","alias_value":"JDOQK6OPSZI4BGT3","created_at":"2026-07-05T10:03:11.734322+00:00"},{"alias_kind":"pith_short_8","alias_value":"JDOQK6OP","created_at":"2026-07-05T10:03:11.734322+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.01408","citing_title":"Artificial Intelligence and Misinformation in Art: Can Vision Language Models Judge the Hand or the Machine Behind the Canvas?","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL","json":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL.json","graph_json":"https://pith.science/api/pith-number/JDOQK6OPSZI4BGT3T5LMU46NTL/graph.json","events_json":"https://pith.science/api/pith-number/JDOQK6OPSZI4BGT3T5LMU46NTL/events.json","paper":"https://pith.science/paper/JDOQK6OP"},"agent_actions":{"view_html":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL","download_json":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL.json","view_paper":"https://pith.science/paper/JDOQK6OP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11628&json=true","fetch_graph":"https://pith.science/api/pith-number/JDOQK6OPSZI4BGT3T5LMU46NTL/graph.json","fetch_events":"https://pith.science/api/pith-number/JDOQK6OPSZI4BGT3T5LMU46NTL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL/action/storage_attestation","attest_author":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL/action/author_attestation","sign_citation":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL/action/citation_signature","submit_replication":"https://pith.science/pith/JDOQK6OPSZI4BGT3T5LMU46NTL/action/replication_record"}},"created_at":"2026-07-05T10:03:11.734322+00:00","updated_at":"2026-07-05T10:03:11.734322+00:00"}