{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IOI27DY5CFSHEN3OAMB5UYEOTT","short_pith_number":"pith:IOI27DY5","schema_version":"1.0","canonical_sha256":"4391af8f1d116472376e0303da608e9cf82fafe603a9c3738fc3a813cdd36d9b","source":{"kind":"arxiv","id":"2408.11119","version":2},"attestation_state":"computed","paper":{"title":"Mistral-SPLADE: LLMs for better Learned Sparse Retrieval","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Jaydeep Sen, Meet Doshi, Rudra Murthy, Vignesh P, Vishwajeet Kumar","submitted_at":"2024-08-20T18:21:54Z","abstract_excerpt":"Learned Sparse Retrievers (LSR) have evolved into an effective retrieval strategy that can bridge the gap between traditional keyword-based sparse retrievers and embedding-based dense retrievers. At its core, learned sparse retrievers try to learn the most important semantic keyword expansions from a query and/or document which can facilitate better retrieval with overlapping keyword expansions. LSR like SPLADE has typically been using encoder only models with MLM (masked language modeling) style objective in conjunction with known ways of retrieval performance improvement such as hard negativ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11119","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.IR","submitted_at":"2024-08-20T18:21:54Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"da318816933cd44160cfb782ca7f271e5c8371ade3752bde3cf09dab3b67de0c","abstract_canon_sha256":"62136de61ad5973a711980b8de7efaa938bde0bb27ced7528cd16936de9bc4c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:59.303431Z","signature_b64":"q6EVtGHg2VUxEPNqUbb8eRSM6ViYx9s0YQ4sZsqCCK70EV4KKcxNrOTlPs03IruGnG3V21ZWV2XFDJ5X+2VwCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4391af8f1d116472376e0303da608e9cf82fafe603a9c3738fc3a813cdd36d9b","last_reissued_at":"2026-07-05T08:57:59.303031Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:59.303031Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mistral-SPLADE: LLMs for better Learned Sparse Retrieval","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Jaydeep Sen, Meet Doshi, Rudra Murthy, Vignesh P, Vishwajeet Kumar","submitted_at":"2024-08-20T18:21:54Z","abstract_excerpt":"Learned Sparse Retrievers (LSR) have evolved into an effective retrieval strategy that can bridge the gap between traditional keyword-based sparse retrievers and embedding-based dense retrievers. At its core, learned sparse retrievers try to learn the most important semantic keyword expansions from a query and/or document which can facilitate better retrieval with overlapping keyword expansions. LSR like SPLADE has typically been using encoder only models with MLM (masked language modeling) style objective in conjunction with known ways of retrieval performance improvement such as hard negativ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11119","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11119/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11119","created_at":"2026-07-05T08:57:59.303084+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11119v2","created_at":"2026-07-05T08:57:59.303084+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11119","created_at":"2026-07-05T08:57:59.303084+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOI27DY5CFSH","created_at":"2026-07-05T08:57:59.303084+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOI27DY5CFSHEN3O","created_at":"2026-07-05T08:57:59.303084+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOI27DY5","created_at":"2026-07-05T08:57:59.303084+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29498","citing_title":"Mask the Target: A Plug-and-Play Regularizer Against LoRA Forgetting","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT","json":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT.json","graph_json":"https://pith.science/api/pith-number/IOI27DY5CFSHEN3OAMB5UYEOTT/graph.json","events_json":"https://pith.science/api/pith-number/IOI27DY5CFSHEN3OAMB5UYEOTT/events.json","paper":"https://pith.science/paper/IOI27DY5"},"agent_actions":{"view_html":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT","download_json":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT.json","view_paper":"https://pith.science/paper/IOI27DY5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11119&json=true","fetch_graph":"https://pith.science/api/pith-number/IOI27DY5CFSHEN3OAMB5UYEOTT/graph.json","fetch_events":"https://pith.science/api/pith-number/IOI27DY5CFSHEN3OAMB5UYEOTT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT/action/storage_attestation","attest_author":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT/action/author_attestation","sign_citation":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT/action/citation_signature","submit_replication":"https://pith.science/pith/IOI27DY5CFSHEN3OAMB5UYEOTT/action/replication_record"}},"created_at":"2026-07-05T08:57:59.303084+00:00","updated_at":"2026-07-05T08:57:59.303084+00:00"}