{"as_of":"2026-07-22T01:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:338cb24c78c40747e51171d0383bdd154d30783e70b777847bf86610a2cd0157","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T03:09:51.453839Z","state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-07-21T06:31:05.380196+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-01T07:06:53.318182Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-01T08:55:35.168254Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"cited_work":{"arxiv_id":"2604.19157","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.19157","snapshot_observed_at":"2026-07-01T08:55:35.168254Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","venue":"cs.LG","work_id":"1a508e75-5326-4b50-a328-f152fb2c9af0","year":2026},"citing_paper":{"arxiv_id":"2605.17170","last_updated":"2026-05-16T21:58:28Z","snapshot_observed_at":"2026-07-06T23:28:12.114488Z","submitted_at":"2026-05-16T21:58:28Z","title":"TriAxialKV: Toward Extreme Low-Precision KV-Cache Quantization for Agentic Inference Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T14:32:56.579146Z"},"links":{"cited_paper":"/paper/2604.19157","citing_paper":"/paper/2605.17170"},"observation_digest":"sha256:079ef214acd7dc0898a4edcccb8f640cbefc502caf3a79c6a13808010f1a2ee7","observation_id":"3264530b-d55c-46b6-9809-a799a5cfc4d5","resolution":{"observed_at":"2026-05-20T14:33:21.298670Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"cited_work":{"arxiv_id":"2604.19157","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.19157","snapshot_observed_at":"2026-07-01T08:55:35.168254Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","venue":"cs.LG","work_id":"1a508e75-5326-4b50-a328-f152fb2c9af0","year":2026},"citing_paper":{"arxiv_id":"2605.17757","last_updated":"2026-05-18T02:24:29Z","snapshot_observed_at":"2026-07-06T23:28:45.646975Z","submitted_at":"2026-05-18T02:24:29Z","title":"OSCAR: Offline Spectral Covariance-Aware Rotation for 2-bit KV Cache Quantization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T12:16:07.797702Z"},"links":{"cited_paper":"/paper/2604.19157","citing_paper":"/paper/2605.17757"},"observation_digest":"sha256:e1ddab6b049c660fe9fb7f90b5b44558c7b04c246d521aeac8d48a18f20da6ce","observation_id":"5701b047-9b3a-4cac-af2f-11ccf86a9784","resolution":{"observed_at":"2026-05-20T12:18:16.555876Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"cited_work":{"arxiv_id":"2604.19157","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.19157","snapshot_observed_at":"2026-07-01T08:55:35.168254Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","venue":"cs.LG","work_id":"1a508e75-5326-4b50-a328-f152fb2c9af0","year":2026},"citing_paper":{"arxiv_id":"2606.29708","last_updated":"2026-06-30T03:13:37Z","snapshot_observed_at":"2026-07-07T00:03:41.292432Z","submitted_at":"2026-06-29T02:24:13Z","title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-30T05:37:13.211613Z"},"links":{"cited_paper":"/paper/2604.19157","citing_paper":"/paper/2606.29708"},"observation_digest":"sha256:5fd2052355724be415287a50711758808b085b69e7af9b906828e63eedce9b54","observation_id":"34b8ac7a-a67f-4d1c-a3bb-fd0fd08793b6","resolution":{"observed_at":"2026-06-30T14:04:45.295164Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"cited_work":{"arxiv_id":"2604.19157","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.19157","snapshot_observed_at":"2026-07-01T08:55:35.168254Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","venue":"cs.LG","work_id":"1a508e75-5326-4b50-a328-f152fb2c9af0","year":2026},"citing_paper":{"arxiv_id":"2606.29708","last_updated":"2026-06-30T03:13:37Z","snapshot_observed_at":"2026-07-07T00:03:41.292432Z","submitted_at":"2026-06-29T02:24:13Z","title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-01T07:06:53.318182Z"},"links":{"cited_paper":"/paper/2604.19157","citing_paper":"/paper/2606.29708"},"observation_digest":"sha256:aff2d3c1fee8036fef69b3c96aec4ef3eb83625b2abb66b51d43daeacc1fec31","observation_id":"5b1ca8d8-5672-4340-9fe7-fccf202243d0","resolution":{"observed_at":"2026-07-01T08:55:35.169438Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2604.19157/citation-record","integrity":"/paper/2604.19157/integrity","json":"/paper/2604.19157/citation-record.json","paper":"/paper/2604.19157"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints","venue":null,"work_id":"112cb6ff-9974-4d28-89ee-b787681a3f15","year":2023},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:5e395e3271e2ccd18694c093eb337a3562ded486a5591a7a970660f087b5c994","observation_id":"1fc4751d-1b7d-4c8a-8a95-2813420122f9","resolution":{"observed_at":"2026-05-22T18:05:01.676258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02153","last_updated":"2025-09-15T22:15:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T18:35:16Z","title":"Small Language Models are the Future of Agentic AI","version":2},"cited_work":{"arxiv_id":"2506.02153","doi":"10.48550/arxiv.2506.02153","metadata_source":"pith","pith_arxiv_id":"2506.02153","snapshot_observed_at":"2026-07-11T01:57:50.210796Z","title":"Small Language Models are the Future of Agentic AI","venue":"cs.AI","work_id":"ba0f0305-4a51-48fd-a13f-201439a18f9e","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2506.02153","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:2117465e5826cff67aaf03b16d89f159cce0d053d2e48ca90f7abd2b2b1f37cd","observation_id":"f460f266-c1ff-4e33-ab90-f3182d1b90e5","resolution":{"observed_at":"2026-05-16T11:55:51.260368Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-10T01:20:11.837567+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T01:20:11.837567+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-07-11T03:27:46.202228Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:996d2bb2f7a9ecdafcd44849337037dbc9db039a50c58fb49865e3b68c55254f","observation_id":"c4b094ad-1f6e-4d08-acbc-fe61eb2c4a6f","resolution":{"observed_at":"2026-05-10T03:14:08.305431Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-11T10:18:59.594684+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T10:18:59.594684+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:52c72fc87478a92aed1931f1d030132210fd55dbc919016e18d5c8e3481e4e2d","observation_id":"a7bacdf5-d0eb-4f28-ad29-f829ca98612c","resolution":{"observed_at":"2026-05-10T03:14:08.300321Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06118","last_updated":"2024-09-11T07:48:26Z","snapshot_observed_at":"2026-07-06T17:14:28.456108Z","submitted_at":"2024-01-11T18:54:44Z","title":"Extreme Compression of Large Language Models via Additive Quantization","version":4},"cited_work":{"arxiv_id":"2401.06118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06118","snapshot_observed_at":"2026-07-04T14:39:57.850108Z","title":"Elias Frantar and Dan Alistarh","venue":null,"work_id":"38bf2888-df57-4a97-a9ca-5d4d524ea463","year":2024},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2401.06118","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:b4a2f22012f7b365a12c3d993712a8a12077f4e7d8e7de5064217cf707f9b70d","observation_id":"0dbeb0ac-ca5b-4fbc-8c29-b34318db1bef","resolution":{"observed_at":"2026-05-10T03:14:08.292383Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.17323","last_updated":"2023-03-22T13:10:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-31T13:42:40Z","title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers","version":2},"cited_work":{"arxiv_id":"2210.17323","doi":"10.48550/arxiv.2210.17323","metadata_source":"pith","pith_arxiv_id":"2210.17323","snapshot_observed_at":"2026-07-10T14:47:14.534663Z","title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers","venue":"cs.LG","work_id":"19ed8c44-202a-48f6-8169-637d5a5f2408","year":2022},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2210.17323","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:67e92966fb1c99c7132b2805ea03cf876bc622a949dc644ac7831337c71ea1b9","observation_id":"ef201b2f-224a-428c-9713-7d9d694b6f01","resolution":{"observed_at":"2026-05-10T17:18:35.295691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-17T20:22:03.028003+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-17T20:22:03.028003+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-07-10T16:57:24.565388Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:18e8ee00c7972948a10a305726258a8c0967d12ac66a3b1f6d6bd8ab46207157","observation_id":"fefe4adf-a0c3-466a-8ea5-efce1be0a46d","resolution":{"observed_at":"2026-05-10T13:00:39.484061Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":"2403.07974","doi":"10.1109/icsme52107.2021.00025","metadata_source":"pith","pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","venue":"cs.SE","work_id":"ea9e51ce-1e75-4182-92d8-4d25f70d2ee4","year":2024},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:42fb21b2ce7f459857fdeff3db3c7f4e5b44213de8542840e89ba4ce872a74f3","observation_id":"6b836c25-2ba9-4800-ab23-4d090d932d1e","resolution":{"observed_at":"2026-05-10T17:34:43.229714Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10169","last_updated":"2023-07-19T17:55:13Z","snapshot_observed_at":"2026-07-06T15:55:59.949673Z","submitted_at":"2023-07-19T17:55:13Z","title":"Challenges and Applications of Large Language Models","version":1},"cited_work":{"arxiv_id":"2307.10169","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10169","snapshot_observed_at":"2026-07-04T09:59:45.738392Z","title":"Challenges and applications of large language models","venue":null,"work_id":"60109b04-1376-4dfc-9989-553108d93b78","year":2023},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2307.10169","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:3ecc2727dd9292669f733a9074db0e20969191155df97823573de1f93d5e85cd","observation_id":"aea58b42-45b8-479f-9d0f-0ffb9a29cf81","resolution":{"observed_at":"2026-05-10T03:14:08.308026Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05527","last_updated":"2024-09-30T22:44:58Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:48:30Z","title":"GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM","version":4},"cited_work":{"arxiv_id":"2403.05527","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.05527","snapshot_observed_at":"2026-07-10T01:36:44.078109Z","title":"Gear: An efficient kv cache compression recipe for near-lossless generative inference of llm","venue":"cs.LG","work_id":"26d3dd66-4e37-4f27-a590-526883aef922","year":2024},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2403.05527","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:ebf60cf75fb55b99a0cd19a953a56914e7128ef7b67c5fa962ec1bd1d999f823","observation_id":"8196466f-24c9-400c-b563-0b79058b682c","resolution":{"observed_at":"2026-05-10T03:14:08.289804Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.18879","last_updated":"2025-06-23T17:50:11Z","snapshot_observed_at":"2026-07-06T21:46:30.250293Z","submitted_at":"2025-06-23T17:50:11Z","title":"CommVQ: Commutative Vector Quantization for KV Cache Compression","version":1},"cited_work":{"arxiv_id":"2506.18879","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.18879","snapshot_observed_at":"2026-07-02T22:57:26.279666Z","title":"Li, J., Zhang, Y ., Hassan, M","venue":null,"work_id":"32b03174-efd6-4f84-949d-6ec263a707dd","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2506.18879","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:23fffb55f286ae3d7bfa334834ce312876577778a728470948dcf5c0f78cc0dd","observation_id":"b2271486-0af3-487c-b0b0-404416548090","resolution":{"observed_at":"2026-05-10T03:14:08.310691Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"02f81378-1e6d-407d-9172-b940510e902d","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:4db6a55334780bc30cbc5391547970d1e01b2a00ea78da70ef280c2c2981dce5","observation_id":"a269f207-fc79-4bc9-9885-ea6962de1d77","resolution":{"observed_at":"2026-05-22T18:05:01.668836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-07-06T16:00:46.542753Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":"2307.16789","doi":"10.48550/arxiv.2307.16789","metadata_source":"pith","pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-07-10T13:37:06.789691Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","venue":"cs.AI","work_id":"3c555b48-a4d9-42dd-9fdd-0f6018fbe9cb","year":2023},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:77bd49e8b3f9fec2287571c7376b298aa2d9f8e0314e42b0cd59d7f7807b7bdb","observation_id":"02801590-f8a8-4050-a820-085c98a8088d","resolution":{"observed_at":"2026-05-10T03:14:08.320530Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:33.730697+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:33.730697+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Eigen attention: Attention in low-rank space for KV cache compression","venue":null,"work_id":"673b077a-c7a8-426b-94aa-83f7d211accc","year":2024},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:a90afab245555695d5548e30e6d9a441d6a3b18fc167041469e9b0f601246ac3","observation_id":"428a8b2e-970b-4868-87d5-f10b13a74b04","resolution":{"observed_at":"2026-05-22T18:05:01.672591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.findings-emnlp.899","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T00:17:46.922699Z","title":"doi: 10.18653/v1/2024.findings-emnlp.899","venue":null,"work_id":"0839fc73-834a-4b4a-a9a2-bb00dfb606e0","year":2024},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:8a428964ef1bde95410aced6427bb992ccaca208b9f5f7a2d56bcb9354a112af","observation_id":"8bdd13dc-c879-463a-aa79-1e6dd9d912e4","resolution":{"observed_at":"2026-05-10T03:14:07.459764Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:12.775189+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:12.775189+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al","venue":null,"work_id":"acc53586-a187-49e5-8fb8-403ac7948eed","year":2017},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:5e4b87959a7f8b6d72e81055dce2f301f5c45306e63d77b378760aaa8a03f8d6","observation_id":"276d13e8-a14d-4186-8b71-1d60016fb61b","resolution":{"observed_at":"2026-05-22T18:05:01.674363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02958","last_updated":"2026-05-05T23:46:52Z","snapshot_observed_at":"2026-07-06T22:44:14.347235Z","submitted_at":"2026-02-03T00:54:32Z","title":"Quant VideoGen: Auto-Regressive Long Video Generation via 2-Bit KV-Cache Quantization","version":5},"cited_work":{"arxiv_id":"2602.02958","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.02958","snapshot_observed_at":"2026-06-29T22:54:01.098907Z","title":"Quant VideoGen: Auto-Regressive Long Video Generation via 2-Bit KV-Cache Quantization","venue":"cs.LG","work_id":"c4d913d9-457a-4993-bc3f-f7857ce5b9b1","year":2026},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2602.02958","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:4bdfbd3ac81a12d33669f1d09cefa3b57400c376a8ea286a974f32593713f868","observation_id":"a3927740-fa61-46d0-ad94-a19809036269","resolution":{"observed_at":"2026-05-10T03:14:08.323070Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.18643","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T05:50:43.587322Z","title":"Accurate and efficient 2-bit kv cache quantization with dynamic channel-wise precision boost","venue":null,"work_id":"10dfb0e4-e4e0-4484-af8c-ad3fbed63dce","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:b6c3f863f7f91e8bca4970417091885e4045a0fdb3ade2ec3bfea531e6da4452","observation_id":"3a621870-3670-40ae-9e51-78081747bd6b","resolution":{"observed_at":"2026-05-10T03:14:08.313255Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:f32ce46ba3c802b8ce37619b28b4e5cd388b9935555c59afc2f1d4b8f7c56af2","observation_id":"57c7eb06-e0a0-472b-97f3-d188dbf1ce25","resolution":{"observed_at":"2026-05-10T03:14:08.315559Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19874","last_updated":"2025-04-28T15:05:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T15:05:35Z","title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","version":1},"cited_work":{"arxiv_id":"2504.19874","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.19874","snapshot_observed_at":"2026-07-09T17:46:25.912940Z","title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","venue":"cs.LG","work_id":"1356ba41-d62f-4cb4-9bed-c69836b5968a","year":2025},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"cited_paper":"/paper/2504.19874","citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:ec5f6915bf076a430c549b60e39e8f420088cdea5a844cd6a90686efa091acc7","observation_id":"f87840a7-2d24-4dce-933d-851e85e5026e","resolution":{"observed_at":"2026-05-20T08:09:22.519591Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"13 Preprint","venue":null,"work_id":"00662560-3fdc-499e-bd43-4c5b6b120675","year":2026},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:34479bda1bfd4152032129fc27f37e68d2062577e05c7a08e6f4e496402a49b7","observation_id":"8ad863f4-fc04-4a61-b2af-7a9521f986ee","resolution":{"observed_at":"2026-05-22T18:05:01.665105Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(K)” and “(K&V)","venue":null,"work_id":"552594c0-ed5a-46a1-a0a6-3ea0833b7aa6","year":2048},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:f36bffc09e290bdd6d800964cc43948e73342422857e0cf63b45d8fe33c29015","observation_id":"a90c3243-2111-423a-9a76-6edfb4764912","resolution":{"observed_at":"2026-05-22T18:05:01.670806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"These two metrics can diverge significantly under memory-pressure conditions, as we explain below","venue":null,"work_id":"6f1ef6ec-fb47-42a4-9b1e-f8d9467947a5","year":2000},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:8c29f3b76d5abb358e2559f812f43f052d7902d9db4bf6db0c6fd184e06d1b48","observation_id":"c145495f-f2e2-43b1-b9c9-1bd97862e92a","resolution":{"observed_at":"2026-05-22T18:05:01.678233Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Under review","venue":null,"work_id":"775ff092-9622-49db-ab31-a46a15cf0558","year":2048},"citing_paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T03:09:51.453839Z"},"links":{"citing_paper":"/paper/2604.19157"},"observation_digest":"sha256:d2023e3076c50390e21eaa9fc88b2ad9bdce2ac62dc89b38981e55c12fdc1d6d","observation_id":"3236f2e8-e37f-4425-b44a-fe3e6ab5a63d","resolution":{"observed_at":"2026-05-22T18:05:01.666830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-21T06:31:05.380196+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.19157","last_updated":"2026-04-21T07:12:23Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:12:23Z","title":"SAW-INT4: System-Aware 4-Bit KV-Cache Quantization for Real-World LLM Serving"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":2,"metadata_mismatch":8,"parse_uncertain":0,"unresolved":0,"verified_exact":8,"verified_fuzzy":6},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-07-21T06:31:05.380196+00:00","source":"crossref"},{"observed_at":"2026-07-21T06:31:00.184556+00:00","source":"retraction_watch"}],"thesis":"As of 22 July 2026, this Paper Citation Record lists 24 of 24 outbound references and 4 inbound Pith citation observations for arXiv:2604.19157."}