{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VE4B3TARJIXU2IXFO62T4HDAXN","short_pith_number":"pith:VE4B3TAR","schema_version":"1.0","canonical_sha256":"a9381dcc114a2f4d22e577b53e1c60bb6515687705a69593b4e12050efddd386","source":{"kind":"arxiv","id":"2502.08246","version":2},"attestation_state":"computed","paper":{"title":"Inference-time sparse attention with asymmetric indexing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Francisco Massa, Gergely Szilvasy, Herv\\'e J\\'egou, Maria Lomeli, Matthijs Douze, Naila Murray, Pierre-Emmanuel Mazar\\'e","submitted_at":"2025-02-12T09:39:54Z","abstract_excerpt":"Self-attention in transformer models is an incremental associative memory that maps key vectors to value vectors. One way to speed up self-attention is to employ GPU-compatible vector search algorithms based on standard partitioning methods such as k-means. However, such partitioning methods yield poor results in this context because (1) the keys and queries follow different distributions, and (2) the RoPE positional encoding hinders the bucket assignment.\n  This paper introduces Saap (Self-Attention with Asymmetric Partitions), which overcomes these problems. It is an asymmetrical indexing te"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08246","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-12T09:39:54Z","cross_cats_sorted":[],"title_canon_sha256":"fe1f11b7b16b3fa04aa4d12260d2f19194765ae0e5b8f11d6c54604b35db5d9b","abstract_canon_sha256":"325d119d8f7955cd50bfd181bdeacf1a8eb27e4640db65fc3db07d0ef04dc645"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:53.565020Z","signature_b64":"YtDrLcHW7bq2tSCqiD5MmDG/ROVwmwqx9a2yDafwktHuvnMtl7Za4N8PiGSzwhmY07LQdR8eChkBORiXyDC+AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a9381dcc114a2f4d22e577b53e1c60bb6515687705a69593b4e12050efddd386","last_reissued_at":"2026-07-05T11:14:53.564595Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:53.564595Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inference-time sparse attention with asymmetric indexing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Francisco Massa, Gergely Szilvasy, Herv\\'e J\\'egou, Maria Lomeli, Matthijs Douze, Naila Murray, Pierre-Emmanuel Mazar\\'e","submitted_at":"2025-02-12T09:39:54Z","abstract_excerpt":"Self-attention in transformer models is an incremental associative memory that maps key vectors to value vectors. One way to speed up self-attention is to employ GPU-compatible vector search algorithms based on standard partitioning methods such as k-means. However, such partitioning methods yield poor results in this context because (1) the keys and queries follow different distributions, and (2) the RoPE positional encoding hinders the bucket assignment.\n  This paper introduces Saap (Self-Attention with Asymmetric Partitions), which overcomes these problems. It is an asymmetrical indexing te"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08246","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08246/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08246","created_at":"2026-07-05T11:14:53.564655+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08246v2","created_at":"2026-07-05T11:14:53.564655+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08246","created_at":"2026-07-05T11:14:53.564655+00:00"},{"alias_kind":"pith_short_12","alias_value":"VE4B3TARJIXU","created_at":"2026-07-05T11:14:53.564655+00:00"},{"alias_kind":"pith_short_16","alias_value":"VE4B3TARJIXU2IXF","created_at":"2026-07-05T11:14:53.564655+00:00"},{"alias_kind":"pith_short_8","alias_value":"VE4B3TAR","created_at":"2026-07-05T11:14:53.564655+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17147","citing_title":"ScenarioControl: Vision-Language Controllable Vectorized Latent Scenario Generation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN","json":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN.json","graph_json":"https://pith.science/api/pith-number/VE4B3TARJIXU2IXFO62T4HDAXN/graph.json","events_json":"https://pith.science/api/pith-number/VE4B3TARJIXU2IXFO62T4HDAXN/events.json","paper":"https://pith.science/paper/VE4B3TAR"},"agent_actions":{"view_html":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN","download_json":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN.json","view_paper":"https://pith.science/paper/VE4B3TAR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08246&json=true","fetch_graph":"https://pith.science/api/pith-number/VE4B3TARJIXU2IXFO62T4HDAXN/graph.json","fetch_events":"https://pith.science/api/pith-number/VE4B3TARJIXU2IXFO62T4HDAXN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN/action/storage_attestation","attest_author":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN/action/author_attestation","sign_citation":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN/action/citation_signature","submit_replication":"https://pith.science/pith/VE4B3TARJIXU2IXFO62T4HDAXN/action/replication_record"}},"created_at":"2026-07-05T11:14:53.564655+00:00","updated_at":"2026-07-05T11:14:53.564655+00:00"}