{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HIBUTUQXU4NNS6CVHDCEYXH6G3","short_pith_number":"pith:HIBUTUQX","schema_version":"1.0","canonical_sha256":"3a0349d217a71ad9785538c44c5cfe36fb742c720baf51508f299f1c362ec8b9","source":{"kind":"arxiv","id":"2501.14205","version":1},"attestation_state":"computed","paper":{"title":"Serving Long-Context LLMs at the Mobile Edge: Test-Time Reinforcement Learning-based Model Caching and Inference Offloading","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.NI","authors_text":"Christopher G. Brinton, Dusit Niyato, Minrui Xu","submitted_at":"2025-01-24T03:21:20Z","abstract_excerpt":"Large Language Models (LLMs) can perform zero-shot learning on unseen tasks and few-shot learning on complex reasoning tasks. However, resource-limited mobile edge networks struggle to support long-context LLM serving for LLM agents during multi-round interactions with users. Unlike stateless computation offloading and static service offloading in edge computing, optimizing LLM serving at edge servers is challenging because LLMs continuously learn from context which raises accuracy, latency, and resource consumption dynamics. In this paper, we propose a joint model caching and inference offloa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14205","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.NI","submitted_at":"2025-01-24T03:21:20Z","cross_cats_sorted":[],"title_canon_sha256":"6a732f56f340ef8329248d1458229e14dd43fa2d7afd0b57f42c7b45ccd2e7ff","abstract_canon_sha256":"5559450ee702375fe65b1e3f3320c4d642e172aab422d3f68482de1eba9671c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:47.183502Z","signature_b64":"3ErSL10T6lVif+1KzoroFVv297Boo/0Rh2IzRqmKjb2rDiDxSPO/QKnobgYmnrd8lZV/xnySsTnMSiUucAmlAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a0349d217a71ad9785538c44c5cfe36fb742c720baf51508f299f1c362ec8b9","last_reissued_at":"2026-07-05T10:04:47.182965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:47.182965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Serving Long-Context LLMs at the Mobile Edge: Test-Time Reinforcement Learning-based Model Caching and Inference Offloading","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.NI","authors_text":"Christopher G. Brinton, Dusit Niyato, Minrui Xu","submitted_at":"2025-01-24T03:21:20Z","abstract_excerpt":"Large Language Models (LLMs) can perform zero-shot learning on unseen tasks and few-shot learning on complex reasoning tasks. However, resource-limited mobile edge networks struggle to support long-context LLM serving for LLM agents during multi-round interactions with users. Unlike stateless computation offloading and static service offloading in edge computing, optimizing LLM serving at edge servers is challenging because LLMs continuously learn from context which raises accuracy, latency, and resource consumption dynamics. In this paper, we propose a joint model caching and inference offloa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14205","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14205/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14205","created_at":"2026-07-05T10:04:47.183026+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14205v1","created_at":"2026-07-05T10:04:47.183026+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14205","created_at":"2026-07-05T10:04:47.183026+00:00"},{"alias_kind":"pith_short_12","alias_value":"HIBUTUQXU4NN","created_at":"2026-07-05T10:04:47.183026+00:00"},{"alias_kind":"pith_short_16","alias_value":"HIBUTUQXU4NNS6CV","created_at":"2026-07-05T10:04:47.183026+00:00"},{"alias_kind":"pith_short_8","alias_value":"HIBUTUQX","created_at":"2026-07-05T10:04:47.183026+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17081","citing_title":"The Price of Anarchy in Disaggregated Inference","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3","json":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3.json","graph_json":"https://pith.science/api/pith-number/HIBUTUQXU4NNS6CVHDCEYXH6G3/graph.json","events_json":"https://pith.science/api/pith-number/HIBUTUQXU4NNS6CVHDCEYXH6G3/events.json","paper":"https://pith.science/paper/HIBUTUQX"},"agent_actions":{"view_html":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3","download_json":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3.json","view_paper":"https://pith.science/paper/HIBUTUQX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14205&json=true","fetch_graph":"https://pith.science/api/pith-number/HIBUTUQXU4NNS6CVHDCEYXH6G3/graph.json","fetch_events":"https://pith.science/api/pith-number/HIBUTUQXU4NNS6CVHDCEYXH6G3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3/action/storage_attestation","attest_author":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3/action/author_attestation","sign_citation":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3/action/citation_signature","submit_replication":"https://pith.science/pith/HIBUTUQXU4NNS6CVHDCEYXH6G3/action/replication_record"}},"created_at":"2026-07-05T10:04:47.183026+00:00","updated_at":"2026-07-05T10:04:47.183026+00:00"}