{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ODXD7XOHSEAPS4PMZK56IP2XTN","short_pith_number":"pith:ODXD7XOH","schema_version":"1.0","canonical_sha256":"70ee3fddc79100f971eccabbe43f579b71cb797a8165b76da4a22f9304ebc3cb","source":{"kind":"arxiv","id":"2406.02616","version":5},"attestation_state":"computed","paper":{"title":"Adaptive Layer Splitting for Wireless LLM Inference in Edge Computing: A Model-Based Reinforcement Learning Approach","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Honggang Zhang, Rongpeng Li, Xiaoxue Yu, Yuxuan Chen, Zhifeng Zhao","submitted_at":"2024-06-03T09:41:42Z","abstract_excerpt":"Optimizing the deployment of large language models (LLMs) in edge computing environments is critical for enhancing privacy and computational efficiency. Toward efficient wireless LLM inference in edge computing, this study comprehensively analyzes the impact of different splitting points in mainstream open-source LLMs. On this basis, this study introduces a framework taking inspiration from model-based reinforcement learning (MBRL) to determine the optimal splitting point across the edge and user equipment (UE). By incorporating a reward surrogate model, our approach significantly reduces the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02616","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-03T09:41:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4329601be7f4525c410f833f1bcc01237525d557432f46cb18038810aef3278c","abstract_canon_sha256":"bf3950481bf2fbf8bb97b3f3618e18f9c5cf5f2824d3ec4c3cf1ea7759efb8e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:36.446895Z","signature_b64":"VbDXmyKZy+JRBKBr8gn8eduOaDWPw6KlguLYVtrRGT65YB8bLttkwsGcitBBIlxOJKD9At6VBnnzw6AlBbGMAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70ee3fddc79100f971eccabbe43f579b71cb797a8165b76da4a22f9304ebc3cb","last_reissued_at":"2026-07-05T09:05:36.446459Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:36.446459Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Layer Splitting for Wireless LLM Inference in Edge Computing: A Model-Based Reinforcement Learning Approach","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Honggang Zhang, Rongpeng Li, Xiaoxue Yu, Yuxuan Chen, Zhifeng Zhao","submitted_at":"2024-06-03T09:41:42Z","abstract_excerpt":"Optimizing the deployment of large language models (LLMs) in edge computing environments is critical for enhancing privacy and computational efficiency. Toward efficient wireless LLM inference in edge computing, this study comprehensively analyzes the impact of different splitting points in mainstream open-source LLMs. On this basis, this study introduces a framework taking inspiration from model-based reinforcement learning (MBRL) to determine the optimal splitting point across the edge and user equipment (UE). By incorporating a reward surrogate model, our approach significantly reduces the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02616","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02616","created_at":"2026-07-05T09:05:36.446520+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02616v5","created_at":"2026-07-05T09:05:36.446520+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02616","created_at":"2026-07-05T09:05:36.446520+00:00"},{"alias_kind":"pith_short_12","alias_value":"ODXD7XOHSEAP","created_at":"2026-07-05T09:05:36.446520+00:00"},{"alias_kind":"pith_short_16","alias_value":"ODXD7XOHSEAPS4PM","created_at":"2026-07-05T09:05:36.446520+00:00"},{"alias_kind":"pith_short_8","alias_value":"ODXD7XOH","created_at":"2026-07-05T09:05:36.446520+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22496","citing_title":"Enabling Cloud-Level Accuracy in Edge AI through IoT Data Preprocessing","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23158","citing_title":"What Does the Server See? Understanding Privacy Leakage from Large Language Models in Split Inference","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2601.11652","citing_title":"WISP: Waste- and Interference-Suppressed Distributed Speculative LLM Serving at the Edge via Dynamic Drafting and SLO-Aware Batching","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN","json":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN.json","graph_json":"https://pith.science/api/pith-number/ODXD7XOHSEAPS4PMZK56IP2XTN/graph.json","events_json":"https://pith.science/api/pith-number/ODXD7XOHSEAPS4PMZK56IP2XTN/events.json","paper":"https://pith.science/paper/ODXD7XOH"},"agent_actions":{"view_html":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN","download_json":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN.json","view_paper":"https://pith.science/paper/ODXD7XOH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02616&json=true","fetch_graph":"https://pith.science/api/pith-number/ODXD7XOHSEAPS4PMZK56IP2XTN/graph.json","fetch_events":"https://pith.science/api/pith-number/ODXD7XOHSEAPS4PMZK56IP2XTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN/action/storage_attestation","attest_author":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN/action/author_attestation","sign_citation":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN/action/citation_signature","submit_replication":"https://pith.science/pith/ODXD7XOHSEAPS4PMZK56IP2XTN/action/replication_record"}},"created_at":"2026-07-05T09:05:36.446520+00:00","updated_at":"2026-07-05T09:05:36.446520+00:00"}