{"as_of":"2026-08-08T20:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b1a5882d2574289b21cc5eeca093dfc392268f0db9cfefc25f67f9cc3bfbf673","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:26:17.238480Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T07:59:39.594987Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2401.05459","last_updated":"2024-05-08T06:16:23Z","snapshot_observed_at":"2026-08-02T13:57:57.119489Z","submitted_at":"2024-01-10T09:25:45Z","title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","version":2},"reference_index":274,"source":"pdf_text","source_observed_at":"2026-05-17T00:57:26.303195Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2401.05459"},"observation_digest":"sha256:850f9bd505d25d9e78e6ee487816cfb784769f8acce81c3db91d5c0d8ab594cf","observation_id":"b0f5232b-ef16-4177-bc9e-752bc21b198a","resolution":{"observed_at":"2026-05-17T00:57:26.808663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":251,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:d724e2f46fe502c0223b57f89ea69d2a327f4a53fde63ea593207cee610d4dde","observation_id":"9bfc00a5-a9a7-4688-9c6e-dacdc19a8d58","resolution":{"observed_at":"2026-05-15T02:39:33.393963Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-11T07:53:38.715353Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2409.19256"},"observation_digest":"sha256:35ace58faf238726ba7f170f2ae0d4c7b81c3944ec9ef318559545d23c736a01","observation_id":"025dc3cd-c52b-4e23-91cd-2d17303e6d9e","resolution":{"observed_at":"2026-05-11T07:53:38.891930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-07T14:26:17.238480Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19235","last_updated":"2025-05-25T17:16:34Z","snapshot_observed_at":"2026-08-07T20:41:54.292300Z","submitted_at":"2025-05-25T17:16:34Z","title":"CoreMatching: A Co-adaptive Sparse Inference Framework with Token and Neuron Pruning for Comprehensive Acceleration of Vision-Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:26:17.238480Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2505.19235"},"observation_digest":"sha256:5572e1143ac86c12c6b841cdc1ff38bd93a3a767a9500d2c2726f1b6c0dc60cd","observation_id":"69b87f83-9793-4573-9514-e99b11547a90","resolution":{"observed_at":"2026-08-07T14:26:17.238480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-07T04:59:17.159468Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09351","last_updated":"2025-06-11T03:05:24Z","snapshot_observed_at":"2026-08-07T20:42:24.080368Z","submitted_at":"2025-06-11T03:05:24Z","title":"DIVE into MoE: Diversity-Enhanced Reconstruction of Large Language Models from Dense into Mixture-of-Experts","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T04:59:17.159468Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2506.09351"},"observation_digest":"sha256:2af7374c31924b18c4a2e20705b3d5fc79defb4b946a94ad95a59d923f33c2a7","observation_id":"8865263a-8f4e-44a8-b067-b61b5be100ea","resolution":{"observed_at":"2026-08-07T04:59:17.159468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-06T23:00:11.173785Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20187","last_updated":"2025-07-02T05:12:29Z","snapshot_observed_at":"2026-08-07T20:42:38.846597Z","submitted_at":"2025-06-25T07:26:42Z","title":"Breaking the Boundaries of Long-Context LLM Inference: Adaptive KV Management on a Single Commodity GPU","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T23:00:11.173785Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2506.20187"},"observation_digest":"sha256:53e4edb56a7307266d475d47fbf714a000dd4fe0dc2bae0dcae540799a734c33","observation_id":"5568e98f-9628-4a2d-b939-7a04aeaab1fb","resolution":{"observed_at":"2026-08-06T23:00:11.173785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-06T18:20:00.384466Z","title":"PowerInfer : Fast large language model serving with a consumer-grade GPU","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08771","last_updated":"2025-07-30T04:14:15Z","snapshot_observed_at":"2026-08-07T06:32:52.508610Z","submitted_at":"2025-07-11T17:28:56Z","title":"BlockFFN: Towards End-Side Acceleration-Friendly Mixture-of-Experts with Chunk-Level Activation Sparsity","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T18:20:00.384466Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2507.08771"},"observation_digest":"sha256:499b2e0baa77541f58d6b4a12103102812288989be1d3a34062cb7ecbbafdc10","observation_id":"0aeb7da2-92c1-4501-bf32-4b3b6ad58c68","resolution":{"observed_at":"2026-08-06T18:20:00.384466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-06T16:57:31.381924Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12205","last_updated":"2025-07-16T13:04:06Z","snapshot_observed_at":"2026-08-07T12:21:15.238746Z","submitted_at":"2025-07-16T13:04:06Z","title":"Toward Efficient SpMV in Sparse LLMs via Block Extraction and Compressed Storage","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T16:57:31.381924Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2507.12205"},"observation_digest":"sha256:63285b51ea290c845288952a3cf28f458b975d94e6e2892c14ddaa58614fde6e","observation_id":"5e860b4a-708f-43e3-aad1-8ed942bdf3a0","resolution":{"observed_at":"2026-08-06T16:57:31.381924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-06T18:17:30.724777Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14179","last_updated":"2025-07-11T19:07:29Z","snapshot_observed_at":"2026-08-07T13:08:31.760400Z","submitted_at":"2025-07-11T19:07:29Z","title":"A Sparsity Predicting Approach for Large Language Models via Activation Pattern Clustering","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T18:17:30.724777Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2507.14179"},"observation_digest":"sha256:b08797f1b29786a3567093d0d23cfe2b4e1fa7332c3028781c0137422b5625e0","observation_id":"2950b28d-ac6b-4d06-9bbf-fcdf52566cca","resolution":{"observed_at":"2026-08-06T18:17:30.724777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2604.17182","last_updated":"2026-04-19T00:56:08Z","snapshot_observed_at":"2026-08-03T00:20:06.527897Z","submitted_at":"2026-04-19T00:56:08Z","title":"Layer-wise MoE Routing Locality under Shared-Prefix Code Generation: Token-Identity Decomposition and Compile-Equivalent Fork Redundancy","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T06:46:52.811371Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2604.17182"},"observation_digest":"sha256:865237b4d87e45565319749fcb6decf963240ae66fa0d5c0e298330693994846","observation_id":"4c185bc3-56b6-4eaa-a43f-8583afd93821","resolution":{"observed_at":"2026-05-10T06:51:46.434711Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2604.27115","last_updated":"2026-04-29T19:08:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-29T19:08:15Z","title":"Exploring the Limits of Pruning: Task-Specific Neurons, Model Collapse, and Recovery in Task-Specific Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-07T10:05:31.211127Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2604.27115"},"observation_digest":"sha256:26f429ea59a19528300222feae4713c6ca11004434588915a93ccd4070f04756","observation_id":"f469afaf-17f4-40c8-abdf-4732a5164f8e","resolution":{"observed_at":"2026-05-12T09:36:27.716071Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2604.27476","last_updated":"2026-06-08T07:15:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:18:50Z","title":"EdgeFM: Efficient Edge Inference for Vision-Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T08:30:35.264804Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2604.27476"},"observation_digest":"sha256:8f32edb499d6e7deb54ba37b378c57417362e58a9507ae3168ba6d0dc6e0b6e8","observation_id":"b3e29c5b-c69b-4ebc-8251-d4c5128ee048","resolution":{"observed_at":"2026-05-12T10:01:27.689914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2604.27476","last_updated":"2026-06-08T07:15:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:18:50Z","title":"EdgeFM: Efficient Edge Inference for Vision-Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-01T08:54:41.845820Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2604.27476"},"observation_digest":"sha256:c25164086880b455dd001bb22083338be18cb60ae044369faee759eab44fa205","observation_id":"0edf9054-3d03-4b0f-b0e9-96ef98c2bd5f","resolution":{"observed_at":"2026-07-01T08:55:34.733096Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2606.12556","last_updated":"2026-06-16T05:18:15Z","snapshot_observed_at":"2026-08-01T05:54:13.696296Z","submitted_at":"2026-06-10T18:06:30Z","title":"ITME: Inference Tiered Memory Expansion with Disaggregated CXL-Hybrid Memories","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T08:11:57.929452Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2606.12556"},"observation_digest":"sha256:30683097bc412a12462111eed2c71fa5d50def7cb6519e5d4f02c90fad7a0920","observation_id":"257da112-d13a-4153-bd07-b2d47146e05d","resolution":{"observed_at":"2026-07-03T13:18:13.435370Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":"2312.12456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-04T07:59:39.594987Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu","venue":null,"work_id":"40f41979-d6fe-4597-a800-fbd226b8a28d","year":2024},"citing_paper":{"arxiv_id":"2606.21428","last_updated":"2026-07-09T17:35:35Z","snapshot_observed_at":"2026-08-08T09:53:30.201517Z","submitted_at":"2026-06-19T13:45:45Z","title":"Does Mixture-of-Experts Actually Help Inference on Consumer and Edge Hardware? An Empirical Study","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-26T12:30:55.628115Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2606.21428"},"observation_digest":"sha256:baad633df4b23be3a259f32f6537d347ea8f417a32788007c8c936c99a9aba28","observation_id":"0f38a8d6-31a3-4311-a2a7-6456ae94e70b","resolution":{"observed_at":"2026-07-04T07:59:39.596383Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-12T13:05:17.273287Z","title":"Alfarizy et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.21428","last_updated":"2026-07-09T17:35:35Z","snapshot_observed_at":"2026-08-08T09:53:30.201517Z","submitted_at":"2026-06-19T13:45:45Z","title":"Does Mixture-of-Experts Actually Help Inference on Consumer and Edge Hardware? An Empirical Study","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-12T13:05:17.273287Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2606.21428"},"observation_digest":"sha256:f22be60966e96994c5f01ea2cda70001003940cb065f44a586173af1ad5a39ed","observation_id":"ef95162c-6e8d-4289-8c31-7a1c19ea98ea","resolution":{"observed_at":"2026-07-12T13:05:17.273287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-07-12T03:14:02.219678Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03333","last_updated":"2026-07-03T13:51:32Z","snapshot_observed_at":"2026-08-04T06:12:58.772470Z","submitted_at":"2026-07-03T13:51:32Z","title":"SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-12T03:14:02.219678Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2607.03333"},"observation_digest":"sha256:4f6da0a6644573fa8a13cd51b66374c156143814e56ce5e75288f506bc0e0cd7","observation_id":"6698b789-2425-45fc-bdd7-130a55fb1c32","resolution":{"observed_at":"2026-07-12T03:14:02.219678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-01T18:04:57.394979Z","title":"Powerinfer: Fast large language model serving with a consumer-grade gpu,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17415","last_updated":"2026-07-19T21:22:22Z","snapshot_observed_at":"2026-08-07T18:11:28.470281Z","submitted_at":"2026-07-19T21:22:22Z","title":"Transition-Aware Backend Dispatch for Edge LLM Inference","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T18:04:57.394979Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2607.17415"},"observation_digest":"sha256:a9d4be9141bfc09936b9985042eb3086b33d959c885c2f3bf242959d1f035221","observation_id":"e6d75a5c-45f6-461b-82ac-cd4a75d5742f","resolution":{"observed_at":"2026-08-01T18:04:57.394979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12456","snapshot_observed_at":"2026-08-01T15:58:29.909466Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.18141","last_updated":"2026-08-05T22:55:21Z","snapshot_observed_at":"2026-08-08T20:14:19.154876Z","submitted_at":"2026-07-20T16:35:47Z","title":"A CXL Memory Rack for Multi-Turn LLM Serving","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-01T15:58:29.909466Z"},"links":{"cited_paper":"/paper/2312.12456","citing_paper":"/paper/2607.18141"},"observation_digest":"sha256:12aaabad7e97abccf34aeee1aacd6bf79cd862e9981617c665546c371e3f5498","observation_id":"dd1d5176-b30e-48b7-8aa1-d750a2da225a","resolution":{"observed_at":"2026-08-01T15:58:29.909466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.12456/citation-record","integrity":"/paper/2312.12456/integrity","json":"/paper/2312.12456/citation-record.json","paper":"/paper/2312.12456"},"outbound":[],"paper":{"arxiv_id":"2312.12456","last_updated":"2024-12-12T12:38:12Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T20:41:48.188566Z","submitted_at":"2023-12-16T02:27:00Z","title":"PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2312.12456."}