{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LZNQEFNKJVTKXR5OYXKCWZ7TLX","short_pith_number":"pith:LZNQEFNK","schema_version":"1.0","canonical_sha256":"5e5b0215aa4d66abc7aec5d42b67f35dead430bc018d4484dbda69359fc3c9d8","source":{"kind":"arxiv","id":"2601.09159","version":4},"attestation_state":"computed","paper":{"title":"LLMs Meet Isolation Kernel: Lightweight, Learning-free Binary Embeddings for Fast Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Isolation Kernel converts LLM embeddings into compact binary codes that deliver up to 16 times lower memory use and 16.7 times faster retrieval with comparable accuracy.","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Cam-Tu Nguyen, Kai Ming Ting, Yang Xu, Zhibo Zhang","submitted_at":"2026-01-14T04:54:09Z","abstract_excerpt":"Large language models (LLMs) have recently enabled remarkable progress in text representation. However, their embeddings are typically high-dimensional, leading to substantial storage and retrieval overhead. Although recent approaches such as Matryoshka Representation Learning (MRL) and Contrastive Sparse Representation (CSR) alleviate these issues to some extent, they still suffer from retrieval accuracy degradation. This paper proposes Isolation Kernel Embedding or IKE, a learning-free method that transforms an LLM embedding into a binary embedding using Isolation Kernel (IK). Lightweight an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.09159","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2026-01-14T04:54:09Z","cross_cats_sorted":[],"title_canon_sha256":"dfecff787b8312f1b2c6cf22ea3ea12d8d7565cf0eb401a4fa16bd44c298a390","abstract_canon_sha256":"f0857584fdc40e90fa2727875040cdfff66670933e7ce17180a984ca41200c29"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-08T01:19:09.565225Z","signature_b64":"vPEQ87MN0U354k23EayofEEO3rd+VPdezPDGbIMGNOAkyr4Hp8pgJCPvgvNx+Xpl1RoogG/UG3ulvsoASGH7BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e5b0215aa4d66abc7aec5d42b67f35dead430bc018d4484dbda69359fc3c9d8","last_reissued_at":"2026-07-08T01:19:09.564661Z","signature_status":"signed_v1","first_computed_at":"2026-07-08T01:19:09.564661Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs Meet Isolation Kernel: Lightweight, Learning-free Binary Embeddings for Fast Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Isolation Kernel converts LLM embeddings into compact binary codes that deliver up to 16 times lower memory use and 16.7 times faster retrieval with comparable accuracy.","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Cam-Tu Nguyen, Kai Ming Ting, Yang Xu, Zhibo Zhang","submitted_at":"2026-01-14T04:54:09Z","abstract_excerpt":"Large language models (LLMs) have recently enabled remarkable progress in text representation. However, their embeddings are typically high-dimensional, leading to substantial storage and retrieval overhead. Although recent approaches such as Matryoshka Representation Learning (MRL) and Contrastive Sparse Representation (CSR) alleviate these issues to some extent, they still suffer from retrieval accuracy degradation. This paper proposes Isolation Kernel Embedding or IKE, a learning-free method that transforms an LLM embedding into a binary embedding using Isolation Kernel (IK). Lightweight an"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Experiments on multiple text retrieval datasets demonstrate that IKE offers up to 16.7x faster retrieval and 16x lower memory usage than the original LLM embeddings, while maintaining comparable accuracy. Theoretically, we show that IKE works because it satisfies four essential criteria for effective binary hashing that other methods do not possess.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the Isolation Kernel applied to LLM embeddings preserves enough semantic similarity information to avoid meaningful accuracy loss on downstream retrieval tasks, without any task-specific tuning or validation of the kernel parameters.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"IKE converts LLM embeddings into binary codes via Isolation Kernel for up to 16.7x faster retrieval and 16x lower memory with comparable accuracy.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Isolation Kernel converts LLM embeddings into compact binary codes that deliver up to 16 times lower memory use and 16.7 times faster retrieval with comparable accuracy.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"c1d6048384e546c73d0f3a460a35542852e10351dbd75b7d34c4e95bef4eb3df"},"source":{"id":"2601.09159","kind":"arxiv","version":4},"verdict":{"id":"b63d16d0-4aec-4ee0-8f51-c43ea41643f1","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-16T15:15:56.325731Z","strongest_claim":"Experiments on multiple text retrieval datasets demonstrate that IKE offers up to 16.7x faster retrieval and 16x lower memory usage than the original LLM embeddings, while maintaining comparable accuracy. Theoretically, we show that IKE works because it satisfies four essential criteria for effective binary hashing that other methods do not possess.","one_line_summary":"IKE converts LLM embeddings into binary codes via Isolation Kernel for up to 16.7x faster retrieval and 16x lower memory with comparable accuracy.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the Isolation Kernel applied to LLM embeddings preserves enough semantic similarity information to avoid meaningful accuracy loss on downstream retrieval tasks, without any task-specific tuning or validation of the kernel parameters.","pith_extraction_headline":"Isolation Kernel converts LLM embeddings into compact binary codes that deliver up to 16 times lower memory use and 16.7 times faster retrieval with comparable accuracy."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.09159/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.09159","created_at":"2026-07-08T01:19:09.564720+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.09159v4","created_at":"2026-07-08T01:19:09.564720+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.09159","created_at":"2026-07-08T01:19:09.564720+00:00"},{"alias_kind":"pith_short_12","alias_value":"LZNQEFNKJVTK","created_at":"2026-07-08T01:19:09.564720+00:00"},{"alias_kind":"pith_short_16","alias_value":"LZNQEFNKJVTKXR5O","created_at":"2026-07-08T01:19:09.564720+00:00"},{"alias_kind":"pith_short_8","alias_value":"LZNQEFNK","created_at":"2026-07-08T01:19:09.564720+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX","json":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX.json","graph_json":"https://pith.science/api/pith-number/LZNQEFNKJVTKXR5OYXKCWZ7TLX/graph.json","events_json":"https://pith.science/api/pith-number/LZNQEFNKJVTKXR5OYXKCWZ7TLX/events.json","paper":"https://pith.science/paper/LZNQEFNK"},"agent_actions":{"view_html":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX","download_json":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX.json","view_paper":"https://pith.science/paper/LZNQEFNK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.09159&json=true","fetch_graph":"https://pith.science/api/pith-number/LZNQEFNKJVTKXR5OYXKCWZ7TLX/graph.json","fetch_events":"https://pith.science/api/pith-number/LZNQEFNKJVTKXR5OYXKCWZ7TLX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX/action/storage_attestation","attest_author":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX/action/author_attestation","sign_citation":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX/action/citation_signature","submit_replication":"https://pith.science/pith/LZNQEFNKJVTKXR5OYXKCWZ7TLX/action/replication_record"}},"created_at":"2026-07-08T01:19:09.564720+00:00","updated_at":"2026-07-08T01:19:09.564720+00:00"}