{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TJ7IDWKT7YGNG367BB5HV7IFQV","short_pith_number":"pith:TJ7IDWKT","schema_version":"1.0","canonical_sha256":"9a7e81d953fe0cd36fdf087a7afd05857c9ebf4366a4bfe8e574a4d3e73c3814","source":{"kind":"arxiv","id":"2407.20272","version":1},"attestation_state":"computed","paper":{"title":"An Efficient Inference Framework for Early-exit Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ruijie Miao, Tong Yang, Xinshuo Yao, Yihan Yan","submitted_at":"2024-07-25T07:50:17Z","abstract_excerpt":"Building efficient inference framework has gained increasing interests for research community. Early-exit models, a variant of LLMs, improves the inference efficiency of LLMs by skipping rest layers and directly generate output tokens when they are confident enough. However, there is no work of LLM inference framework that takes early-exit models into consideration. This is non-trivial as prior art on LLM inference cannot be directly applied to early-exit models. In this work, we solves two key challenges in building efficient inference framework for early-exit models: (1) batch inference at i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20272","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-25T07:50:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ac25f2cb20b0a757d17c2dda0dbf737dc573280557e0e67b8d8a07ab5a8934c9","abstract_canon_sha256":"b0cf4bfaa5c74c3d06963950ab3d9e2d01031e5f1ed7ab286c5cdc3fa6353c70"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:55.358901Z","signature_b64":"YdM5ew5bEi9iiIVPqSZ37u6EW2Pz4X5kCXRqkIcv4gE/Ge1wda6DihztY/oD163YpsNwbpIfZ+fg0LmCDG5kDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a7e81d953fe0cd36fdf087a7afd05857c9ebf4366a4bfe8e574a4d3e73c3814","last_reissued_at":"2026-07-05T08:49:55.358291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:55.358291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Efficient Inference Framework for Early-exit Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ruijie Miao, Tong Yang, Xinshuo Yao, Yihan Yan","submitted_at":"2024-07-25T07:50:17Z","abstract_excerpt":"Building efficient inference framework has gained increasing interests for research community. Early-exit models, a variant of LLMs, improves the inference efficiency of LLMs by skipping rest layers and directly generate output tokens when they are confident enough. However, there is no work of LLM inference framework that takes early-exit models into consideration. This is non-trivial as prior art on LLM inference cannot be directly applied to early-exit models. In this work, we solves two key challenges in building efficient inference framework for early-exit models: (1) batch inference at i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20272","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20272/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20272","created_at":"2026-07-05T08:49:55.358374+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20272v1","created_at":"2026-07-05T08:49:55.358374+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20272","created_at":"2026-07-05T08:49:55.358374+00:00"},{"alias_kind":"pith_short_12","alias_value":"TJ7IDWKT7YGN","created_at":"2026-07-05T08:49:55.358374+00:00"},{"alias_kind":"pith_short_16","alias_value":"TJ7IDWKT7YGNG367","created_at":"2026-07-05T08:49:55.358374+00:00"},{"alias_kind":"pith_short_8","alias_value":"TJ7IDWKT","created_at":"2026-07-05T08:49:55.358374+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06503","citing_title":"Doomed from the Start: Early Abort of LLM Agent Episodes via a Recall-Controlled Probe Cascade","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18396","citing_title":"River-LLM: Large Language Model Seamless Exit Based on KV Share","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18396","citing_title":"River-LLM: Large Language Model Seamless Exit Based on KV Share","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV","json":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV.json","graph_json":"https://pith.science/api/pith-number/TJ7IDWKT7YGNG367BB5HV7IFQV/graph.json","events_json":"https://pith.science/api/pith-number/TJ7IDWKT7YGNG367BB5HV7IFQV/events.json","paper":"https://pith.science/paper/TJ7IDWKT"},"agent_actions":{"view_html":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV","download_json":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV.json","view_paper":"https://pith.science/paper/TJ7IDWKT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20272&json=true","fetch_graph":"https://pith.science/api/pith-number/TJ7IDWKT7YGNG367BB5HV7IFQV/graph.json","fetch_events":"https://pith.science/api/pith-number/TJ7IDWKT7YGNG367BB5HV7IFQV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV/action/storage_attestation","attest_author":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV/action/author_attestation","sign_citation":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV/action/citation_signature","submit_replication":"https://pith.science/pith/TJ7IDWKT7YGNG367BB5HV7IFQV/action/replication_record"}},"created_at":"2026-07-05T08:49:55.358374+00:00","updated_at":"2026-07-05T08:49:55.358374+00:00"}