{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4BCALDEKFOGCLCVVEOG6GYODEA","short_pith_number":"pith:4BCALDEK","schema_version":"1.0","canonical_sha256":"e044058c8a2b8c258ab5238de361c320227dd8e4448df1aaa78e2446dddecf55","source":{"kind":"arxiv","id":"2411.19146","version":5},"attestation_state":"computed","paper":{"title":"Puzzle: Distillation-Based NAS for Inference-Optimized LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akhiad Bercovich, Amnon Geifman, Ehud Karpas, Ido Galil, Ido Shahaf, Itamar Schen, Itay Levy, Izhak Golan, Mohammad Dabbah, Najeeb Nabwani, Nave Assaf, Netanel Haber, Nir Ailon, Omer Ullman Argov, Omri Puny, Oren Tropp, Pavlo Molchanov, Ran El-Yaniv, Ran Rubin, Ran Zilberstein, Roi Koren, Shahar Mor, Talor Abramovich, Tomer Ronen, Yonatan Geifman, Zach Moshe","submitted_at":"2024-11-28T13:45:42Z","abstract_excerpt":"Large language models (LLMs) offer remarkable capabilities, yet their high inference costs restrict wider adoption. While increasing parameter counts improves accuracy, it also broadens the gap between state-of-the-art capabilities and practical deployability. We present Puzzle, a hardware-aware framework that accelerates the inference of LLMs while preserving their capabilities. Using neural architecture search (NAS) at a large-scale, Puzzle optimizes models with tens of billions of parameters. Our approach utilizes blockwise local knowledge distillation (BLD) for parallel architecture explor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.19146","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-28T13:45:42Z","cross_cats_sorted":[],"title_canon_sha256":"298c8a5bb6aca511220512c9a06b775f45b54a872fbf35852d46351427067b46","abstract_canon_sha256":"a96461e31295f904514fb4d38b1edc5755df9e731678da6a498ceec0bb96cb27"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:51.215176Z","signature_b64":"THKMsyRKz4aICzY1UMihzap2W+E/+/CnR16WOPaUWhu9yhFH08G66fjoUyUk7oqFEpY4Ou253lr/PUNJlyyoAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e044058c8a2b8c258ab5238de361c320227dd8e4448df1aaa78e2446dddecf55","last_reissued_at":"2026-07-05T11:14:51.214688Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:51.214688Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Puzzle: Distillation-Based NAS for Inference-Optimized LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akhiad Bercovich, Amnon Geifman, Ehud Karpas, Ido Galil, Ido Shahaf, Itamar Schen, Itay Levy, Izhak Golan, Mohammad Dabbah, Najeeb Nabwani, Nave Assaf, Netanel Haber, Nir Ailon, Omer Ullman Argov, Omri Puny, Oren Tropp, Pavlo Molchanov, Ran El-Yaniv, Ran Rubin, Ran Zilberstein, Roi Koren, Shahar Mor, Talor Abramovich, Tomer Ronen, Yonatan Geifman, Zach Moshe","submitted_at":"2024-11-28T13:45:42Z","abstract_excerpt":"Large language models (LLMs) offer remarkable capabilities, yet their high inference costs restrict wider adoption. While increasing parameter counts improves accuracy, it also broadens the gap between state-of-the-art capabilities and practical deployability. We present Puzzle, a hardware-aware framework that accelerates the inference of LLMs while preserving their capabilities. Using neural architecture search (NAS) at a large-scale, Puzzle optimizes models with tens of billions of parameters. Our approach utilizes blockwise local knowledge distillation (BLD) for parallel architecture explor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.19146","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.19146/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.19146","created_at":"2026-07-05T11:14:51.214745+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.19146v5","created_at":"2026-07-05T11:14:51.214745+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.19146","created_at":"2026-07-05T11:14:51.214745+00:00"},{"alias_kind":"pith_short_12","alias_value":"4BCALDEKFOGC","created_at":"2026-07-05T11:14:51.214745+00:00"},{"alias_kind":"pith_short_16","alias_value":"4BCALDEKFOGCLCVV","created_at":"2026-07-05T11:14:51.214745+00:00"},{"alias_kind":"pith_short_8","alias_value":"4BCALDEK","created_at":"2026-07-05T11:14:51.214745+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24747","citing_title":"Scaling Laws for Task-Specific LLM Distillation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA","json":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA.json","graph_json":"https://pith.science/api/pith-number/4BCALDEKFOGCLCVVEOG6GYODEA/graph.json","events_json":"https://pith.science/api/pith-number/4BCALDEKFOGCLCVVEOG6GYODEA/events.json","paper":"https://pith.science/paper/4BCALDEK"},"agent_actions":{"view_html":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA","download_json":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA.json","view_paper":"https://pith.science/paper/4BCALDEK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.19146&json=true","fetch_graph":"https://pith.science/api/pith-number/4BCALDEKFOGCLCVVEOG6GYODEA/graph.json","fetch_events":"https://pith.science/api/pith-number/4BCALDEKFOGCLCVVEOG6GYODEA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA/action/storage_attestation","attest_author":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA/action/author_attestation","sign_citation":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA/action/citation_signature","submit_replication":"https://pith.science/pith/4BCALDEKFOGCLCVVEOG6GYODEA/action/replication_record"}},"created_at":"2026-07-05T11:14:51.214745+00:00","updated_at":"2026-07-05T11:14:51.214745+00:00"}