{"as_of":"2026-08-13T10:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:67ab98c5537d4220d35187bc8ab60d514952bce1f8af01531fe793d8c8f2eb7f","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T13:46:01.321581Z","state":"measured"},{"denominator":89,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":89,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.15982/citation-record","integrity":"/paper/2411.15982/integrity","json":"/paper/2411.15982/citation-record.json","paper":"/paper/2411.15982"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.916127Z","title":"Resq: Residual quantization for video perception,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.916127Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6f7740947a776f1d07e5e1bf5e9dc0f32c52435c730b7c10ccec83ca59caa5e5","observation_id":"eb908304-59b4-4082-8b4e-8ca13af78c73","resolution":{"observed_at":"2026-08-12T13:46:00.916127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.924633Z","title":"Bit-pragmatic deep neural network computing,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.924633Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:b94129782f38d3590c157d4c49d4947b935d6c9cc45967704b7ff0a312080074","observation_id":"07b6e85e-10eb-4742-ad69-eb44c17f2a83","resolution":{"observed_at":"2026-08-12T13:46:00.924633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.928808Z","title":"Explaining neural scaling laws,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.928808Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:8fdcabdfa0268f1fd081e8383ad99e71d783fd9430009ebf1c71fc31cd08a553","observation_id":"109156cc-ec88-4c3e-b740-788a6f30fc11","resolution":{"observed_at":"2026-08-12T13:46:00.928808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.933063Z","title":"Longbench: A bilingual, multitask benchmark for long context understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.933063Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:defbffecb5791c4e47cba8108ad233f114eecd04004d8811007a23375da0c5f1","observation_id":"f22e992c-6df8-4344-b35e-d24c47398f35","resolution":{"observed_at":"2026-08-12T13:46:00.933063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.937312Z","title":"Demystifying chatgpt: An in-depth survey of openai’s robust large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.937312Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:80a496ba3ad9d1c6291861aea67ce31dc5608dffe978de7fd7866b77c37dd511","observation_id":"c2fbf310-6db7-48f4-b945-3546b592e458","resolution":{"observed_at":"2026-08-12T13:46:00.937312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.941946Z","title":"Genus synthesis solution,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.941946Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:56b78b9a09ae097a025fc0094dce2e0f7fc89c8cc669c80ead59252d7d4b96a5","observation_id":"512cd457-8d26-4023-b16d-7a04113f5a87","resolution":{"observed_at":"2026-08-12T13:46:00.941946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.946229Z","title":"General purpose deep learning accelerator based on bit interleaving,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.946229Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9fe14285911da6662de818e5d455a9b68008a2553aec29e596f3c270ae718fed","observation_id":"5c53dec2-f8da-4aba-a7c7-4e8f421615c9","resolution":{"observed_at":"2026-08-12T13:46:00.946229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.950195Z","title":"Quip: 2-bit quanti- zation of large language models with guarantees,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.950195Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e4a0c51dd00838bbd56b5d85268248daf14c57052c574b32fbc7c18270479b16","observation_id":"f864ed24-45c9-401f-ae54-5c8143377e09","resolution":{"observed_at":"2026-08-12T13:46:00.950195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11062","last_updated":"2025-05-19T06:20:01Z","snapshot_observed_at":"2026-08-12T23:24:47.290027Z","submitted_at":"2024-07-10T17:53:30Z","title":"EfficientQAT: Efficient Quantization-Aware Training for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11062","snapshot_observed_at":"2026-08-12T13:46:00.954375Z","title":"Efficientqat: Efficient quantization-aware training for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.954375Z"},"links":{"cited_paper":"/paper/2407.11062","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:a7b1271f9848d6f53977a092ee584e0ebd7713e4f31ffe5e226b850a189db050","observation_id":"db8e462a-25cb-46fb-b2b5-1b5a23573c3e","resolution":{"observed_at":"2026-08-12T13:46:00.954375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:00.959299Z","title":"Nacl: A general and effective kv cache eviction framework for llm at inference time,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.959299Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:2ddd33053cd8631eb35294abcf4d270e441b352e7cdf66523a0874de2774b97c","observation_id":"c08c524f-0c14-4d61-b5d6-f0409de0d6e1","resolution":{"observed_at":"2026-08-12T13:46:00.959299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.683922Z","title":"Palm: Scaling language modeling with pathways,","venue":null,"work_id":"6989fed4-0dce-4c5d-bb3f-dcbca285a378","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.963367Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:93076b4b7fc8f53a6ce9a706e77b9b5e57b7b5066d0f727cd29119e062f8ac3e","observation_id":"3e140eb0-9c01-4fba-b3db-04b06150bd15","resolution":{"observed_at":"2026-08-12T13:46:02.688552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.668947Z","title":"Vs-quant: Per-vector scaled quantization for accurate low-precision neural network inference,","venue":null,"work_id":"e6ec428c-42eb-49b0-96c3-2022726f2244","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.967701Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:59221bc96cd49bb7ec1188fea7c1f5dc005d701a056f73d7b812cca90cb76fab","observation_id":"0e391695-d4ee-48c0-a897-a30f0f232543","resolution":{"observed_at":"2026-08-12T13:46:02.673582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.655841Z","title":"Pushing the limits of narrow precision inferencing at cloud scale with microsoft floating point,","venue":null,"work_id":"d0ea9865-bddf-4562-9167-0e1caa06eadc","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.971544Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:dd72792b5b9efedbf1916d9c5217c6791bc8332f1c88ae3680456aaaebba5fc1","observation_id":"cbfde174-f710-43b3-bc1e-c0588e604377","resolution":{"observed_at":"2026-08-12T13:46:02.660384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.643539Z","title":"With shared microexponents, a little shifting goes a long way,","venue":null,"work_id":"a061b0c8-15bf-4e57-8ea9-fca663d4d90b","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.976155Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:1e36aa0a8114e270f64547e41d56d325de632adba96eba2dae8116a6e33a12b5","observation_id":"987c2874-d2d8-4184-a3d2-f795bbdf96e0","resolution":{"observed_at":"2026-08-12T13:46:02.647418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.630764Z","title":"A timing-driven approach to synthesize fast barrel shifters,","venue":null,"work_id":"ef69b834-1e62-4204-a5da-ccda12395395","year":2008},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.980180Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:caed685aa3d61728dc4c26030980b8398c01d633063d133e563f5d26a4f13884","observation_id":"d0088e03-3677-4f8a-8050-e4d1ccb8efc8","resolution":{"observed_at":"2026-08-12T13:46:02.635399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.617250Z","title":"Llm.int8(): 8- bit matrix multiplication for transformers at scale,","venue":null,"work_id":"cb4a5ad0-724f-4a4f-b6a3-1115c67304f7","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.984391Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:4f00af0c72ae1540e557f04a2b5785057a966e1ba022df1b377179c5cdbe0238","observation_id":"ec788826-bdd6-4922-b1b5-c725344a1534","resolution":{"observed_at":"2026-08-12T13:46:02.622334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.600868Z","title":"The case for 4-bit precision: k- bit inference scaling laws,","venue":null,"work_id":"efd3de95-d187-4ca0-a777-561280b8b94a","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.988162Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:bdef7ab2d6023d9edc01cfb02d4d5a42dbfec2bdf3691ee371b8e37783254aa1","observation_id":"8480b9a4-4201-4813-a0d8-68b85334e80b","resolution":{"observed_at":"2026-08-12T13:46:02.606453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.582859Z","title":"Hawq: Hessian aware quantization of neural networks with mixed-precision,","venue":null,"work_id":"22b7c12a-b4fc-4b7a-9004-db0e82e44252","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.992180Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:f069af9aa5c82f5f9887056a0fa1302a43112a49e15b06f2a464a0dac4dcb95e","observation_id":"00ecbf6f-8afa-4325-9fb0-89afa02558c9","resolution":{"observed_at":"2026-08-12T13:46:02.588365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.566937Z","title":"Training dnns with hybrid block floating point,","venue":null,"work_id":"ce681c5a-2aeb-4e54-9149-18e733bb6c61","year":2018},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:00.996132Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5b619633201589c505be205f913be7e0b11002822ea9ff37679d8a17dde9917c","observation_id":"e6510551-5932-40cc-b624-1173024cb859","resolution":{"observed_at":"2026-08-12T13:46:02.571178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.553464Z","title":"Skvq: Sliding-window key and value cache quantization for large language models,","venue":null,"work_id":"dfeb37c8-4dea-43d7-ad8b-7bfca1ed6a54","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.000362Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:3739aee2ede19057ce607bc12794979c787dbbbd62a8c3b74b6de75ba16f508c","observation_id":"dce06533-ae2f-40b8-af4b-eab6c87e6b06","resolution":{"observed_at":"2026-08-12T13:46:02.557598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.540840Z","title":"Extreme compression of large language models via additive quantization,","venue":null,"work_id":"90b2ed13-44fe-43ec-a28f-e7728661a0f0","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.004503Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5e14a657b0a7385854b8965af13fcea30be52e809d2884cfda739156018deea2","observation_id":"178aaf15-301b-40f6-8b13-bc2fbe940d64","resolution":{"observed_at":"2026-08-12T13:46:02.545100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.528329Z","title":"Reconfig- urable acceleration of 3d-cnns for human action recognition with block floating-point representation,","venue":null,"work_id":"7dfb0ec6-ff84-4830-af34-3cb65b129f72","year":2018},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.008439Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:10fc552b13725376abbcb445bc7888323530158dec9f8be5d26cded2c1e48c64","observation_id":"f389ac12-1c55-47f5-80ad-99753d11967f","resolution":{"observed_at":"2026-08-12T13:46:02.532602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.516037Z","title":"Static block floating-point quantization for convolutional neural networks on fpga,","venue":null,"work_id":"10a32339-d6a8-4224-9fb4-ef246690af60","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.012376Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:399e8dd0012b025c0b83072413cdc14f8560ff578e5dc21b7980c7ee6ef2c124","observation_id":"01002d37-24f7-4610-8d4a-c2d9c229c290","resolution":{"observed_at":"2026-08-12T13:46:02.520801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.502032Z","title":"Optq: Accurate quantization for generative pre-trained transformers,","venue":null,"work_id":"df79867d-09c4-49cd-a058-aa2d6cdb1165","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.016683Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:c9b6ce69636314f8d8e9ee8dfb10a4e1ab97e936f683cfe99da1eb0ca8061f8d","observation_id":"553aa229-a25a-4769-a67b-c312da294cc8","resolution":{"observed_at":"2026-08-12T13:46:02.506326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.06001","last_updated":"2024-10-09T06:09:41Z","snapshot_observed_at":"2026-08-13T00:11:03.486060Z","submitted_at":"2024-05-09T11:49:05Z","title":"LLMC: Benchmarking Large Language Model Quantization with a Versatile Compression Toolkit","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.06001","snapshot_observed_at":"2026-08-12T13:46:01.020686Z","title":"Llmc: Benchmarking large language model quantization with a versatile compression toolkit,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.020686Z"},"links":{"cited_paper":"/paper/2405.06001","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:1435f00dd723194b086c2aa07638ce3c1cab631789eabf737ca94cb05e5b22ea","observation_id":"be801d61-b0fc-4bb3-a27d-e66b430b3b5c","resolution":{"observed_at":"2026-08-12T13:46:01.020686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.482663Z","title":"Boost: block minifloat-based on-device cnn training accelerator with transfer learning,","venue":null,"work_id":"c56422d3-c614-43f0-8d96-5a9d903f0455","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.025336Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9b2041fd1a7dada8def7db504438ce3d2524e6a637a61c5baf9bf1dc6c258dcb","observation_id":"6fcd9706-878e-4c1c-a67b-2c44d6dc4687","resolution":{"observed_at":"2026-08-12T13:46:02.488289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.468136Z","title":"Olive: Accelerating large language models via hardware- friendly outlier-victim pair quantization,","venue":null,"work_id":"995505cf-b142-4277-82ed-882a51d2b1f1","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.029221Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:2487145d4e000b2e705896adde4acf9ceb1ea5c8956dbd394502c26ded263c9a","observation_id":"2efccf6d-a519-4127-b9b4-af72090c2eac","resolution":{"observed_at":"2026-08-12T13:46:02.472758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.454996Z","title":"Ese: Efficient speech recognition engine with sparse lstm on fpga,","venue":null,"work_id":"283d9bd6-e0f7-438b-b0dd-101efb8b33d0","year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.033280Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:2654028ad46c3a356bb62910fb58cda2a59414108fb69a7e480299494972bf34","observation_id":"734655d4-95bf-4ea4-b953-1923a5d8a9fe","resolution":{"observed_at":"2026-08-12T13:46:02.459389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.18079","last_updated":"2025-05-28T18:58:29Z","snapshot_observed_at":"2026-08-13T04:29:50.330115Z","submitted_at":"2024-01-31T18:58:14Z","title":"KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.18079","snapshot_observed_at":"2026-08-12T13:46:01.037498Z","title":"Kvquant: Towards 10 million context length llm inference with kv cache quantization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.037498Z"},"links":{"cited_paper":"/paper/2401.18079","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:40f56f0abce37dcc8adb472e3cd845293152a7fb753bcd46aab5a39f2e4836e6","observation_id":"e31e43a2-f16c-464b-a096-57ffd31fb592","resolution":{"observed_at":"2026-08-12T13:46:01.037498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.440271Z","title":"A precision-scalable risc-v dnn processor with on-device learning capability at the extreme edge,","venue":null,"work_id":"bf578dd9-306c-455b-a08f-fd145e3e1725","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.042306Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:7c69ab9929b03ee2695af12222606652ce6807b2514bc226d43d1a1a6f38bc5f","observation_id":"acb4a88b-71fd-425a-a4d4-7557df0a3b61","resolution":{"observed_at":"2026-08-12T13:46:02.445583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.424203Z","title":"Mind the gap: Attainable data movement and operational intensity bounds for tensor algorithms,","venue":null,"work_id":"1f9f30dd-abd6-4332-8674-edea1ea0e596","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.046677Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9d1417ce98c280fb334aadb61e3b19b0b8fcd81b7d7b8ef90ad79fee8e238faa","observation_id":"3c4a177a-2705-4842-99c8-b7d2c8a2e8a7","resolution":{"observed_at":"2026-08-12T13:46:02.429805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.404127Z","title":"Figna: Integer unit-based accel- erator design for fp-int gemm preserving numerical accuracy,","venue":null,"work_id":"58a226bb-c343-462d-b873-7abdec494ad9","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.050530Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:67bbcf4bd87e478ca63eb51a45d7dc765d8d8c2d1b3e8fbb8f5b31b47540b624","observation_id":"9feedc6f-68ee-45ee-b614-0cd0da64d635","resolution":{"observed_at":"2026-08-12T13:46:02.410522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.387145Z","title":"Perplexity—a measure of the difficulty of speech recognition tasks,","venue":null,"work_id":"0e8e5695-e332-454c-b565-ce92e62129b3","year":1977},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.054333Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:ae2aab72e35104d063f4396763d9ae814caa2aa02ee23b5d1f98ffbda3b888ea","observation_id":"b971c7e3-66b4-44ab-9994-26903bdfdec2","resolution":{"observed_at":"2026-08-12T13:46:02.392252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.366147Z","title":"Mr. biq: Post-training non- uniform quantization based on minimizing the reconstruction error,","venue":null,"work_id":"1524a179-17f1-4406-b2e5-d34985500d9f","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.060235Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:747e980f5aa05141ce9ffc7a0da6f0f113abfc422462f19fb931238c11558d98","observation_id":"cca7ac48-4bed-454b-91a7-e725e24eecce","resolution":{"observed_at":"2026-08-12T13:46:02.371118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.351155Z","title":"Biqgemm: matrix multiplication with lookup table for binary-coding-based quan- tized dnns,","venue":null,"work_id":"1df7789d-880b-4ae2-90b7-3a19dfc9d87f","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.064416Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:f6ec7fd3694525a67f5939480bad52f30d4922f0af567463a3611bc25a01f612","observation_id":"5a868e5d-84ae-43b1-a764-91759d50d4b7","resolution":{"observed_at":"2026-08-12T13:46:02.356272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.336265Z","title":"Ten lessons from three generations shaped google’s tpuv4i: Industrial product,","venue":null,"work_id":"c9012d61-ddeb-4868-b13f-f0f1de683929","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.069776Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:15fae2901e0df52f115868b42b331703ee0db528eae6a15f53fb88c44ecc829e","observation_id":"28571475-c8a3-430d-97f8-fb952857a7fa","resolution":{"observed_at":"2026-08-12T13:46:02.341385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.321905Z","title":"Stripes: Bit-serial deep neural network computing,","venue":null,"work_id":"669eb79d-6cce-4f5c-9f7d-d91f6fd10eb2","year":2016},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.073787Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:b13ec38d1c778fd8dda37f2d265925bbac33ad1589863cd39e04e02600397218","observation_id":"fe860abe-bc83-47e5-b316-3a520278d183","resolution":{"observed_at":"2026-08-12T13:46:02.326657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.078272Z","title":"A survey of gpt-3 family large language models including chatgpt and gpt-4,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.078272Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:073ca51e8913a3e9a761cb9967ad148e573b539dcb4d0087ab44cb2021000dee","observation_id":"94942890-0859-4f55-b5a3-4a08a3024924","resolution":{"observed_at":"2026-08-12T13:46:01.078272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.299167Z","title":"A 95.6-tops/w deep learning inference accelerator with per-vector scaled 4-bit quantization in 5 nm,","venue":null,"work_id":"50f0bbdd-1a2b-416b-9f36-d1434307ca54","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.082459Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:66b63d9c9814d3cf4029468f9bedc93ef814c92e11d3f16d5e2643628113d5d8","observation_id":"02267acb-faf9-41a0-8f47-18c580f1c56c","resolution":{"observed_at":"2026-08-12T13:46:02.304176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.284082Z","title":"Compressed context mem- ory for online language model interaction,","venue":null,"work_id":"fc5b7f35-fd77-4b8e-b9b8-30f65d424d54","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.086446Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:beb753e0f421a9974f0d4c0864596ff060aa902faa766c30c8e194288b8a4459","observation_id":"e01546cb-3e38-4a57-a7db-f2803705ecdf","resolution":{"observed_at":"2026-08-12T13:46:02.288380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.269127Z","title":"Dacapo: Accelerating continuous learning in autonomous systems for video analytics,","venue":null,"work_id":"cb2ecaff-e4a4-4ea7-b200-e56f1be30970","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.090475Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:891f7b4b470c5177c13f0823caa6dc4fc3e9162f6cd1c2360bfc97bab2d2d198","observation_id":"ba442fe0-bdc1-4a37-be6d-7f271a5796e7","resolution":{"observed_at":"2026-08-12T13:46:02.273986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.255303Z","title":"Winning both the accuracy of floating point activation and the simplicity of integer arithmetic,","venue":null,"work_id":"259f548d-6dd6-42f1-8037-0df7e9a2a870","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.096261Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:62ffb2e57351947331ea4364eff3b1f3b4646127053fd2d3b61abc016ef352cc","observation_id":"048c2a80-8910-4127-b228-ffba59500712","resolution":{"observed_at":"2026-08-12T13:46:02.260399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.239899Z","title":"One-shot model for mixed-precision quantization,","venue":null,"work_id":"b21a974c-0107-43ed-bf83-303c567656c8","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.101397Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:92dc960614cbeebc9e11a0c60e950ec6511b80941b3bf6c973255bf56d76ab9d","observation_id":"05f82623-88e2-48b1-9229-de66136b466e","resolution":{"observed_at":"2026-08-12T13:46:02.244419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.226728Z","title":"Flexpoint: An adaptive numerical format for efficient training of deep neural networks,","venue":null,"work_id":"3f85079d-3bd7-4a04-a969-7d14af7c09f6","year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.106244Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:db3a9168e2b4c6915096586211be805de7020840e748b93f6134a0beb228578a","observation_id":"6de4f875-f11c-4ed3-9f6a-a87b72e63cee","resolution":{"observed_at":"2026-08-12T13:46:02.231230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.214070Z","title":"Tender: Accelerating large language models via tensor decomposition and runtime requantization,","venue":null,"work_id":"8e02cae2-0283-4d73-a37d-07a663ad8bba","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.110394Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5c625648112a7d9a6f7703a3c4ff67f71c8fbef652a35dce2f042ca4a1dc6291","observation_id":"6ff985f6-a23d-486b-8f4d-8d73abf68337","resolution":{"observed_at":"2026-08-12T13:46:02.218324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.201351Z","title":"Bitcluster: Fine-grained weight quantization for load-balanced bit-serial neural network accelerators,","venue":null,"work_id":"b60c5c10-cea7-4e39-a0d1-a85594a42862","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.114610Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:4cd7d185060e4180b22f1bc0c98b8c2d6ad181505acaf47070e9c1cba7debc9e","observation_id":"9c0e6188-8cdf-4159-85a6-5069701f9205","resolution":{"observed_at":"2026-08-12T13:46:02.206101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.185539Z","title":"Norm tweaking: High-performance low-bit quantization of large language models,","venue":null,"work_id":"c1e22cd5-601b-4f41-8da3-d1d1f0a58d54","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.118186Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:7689b1bc6f900c1029d18ee955bd3bf77ab916dbf5fb3f83c5308474fc8f5faf","observation_id":"19bd6321-1c54-4f04-8dc9-116e8de5dff2","resolution":{"observed_at":"2026-08-12T13:46:02.190206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.171845Z","title":"Geo: Generation and execution optimized stochastic computing accelerator for neural networks,","venue":null,"work_id":"72823668-296a-466a-8cbb-4fbd58c2758a","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.122265Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:110f2213ff3e506e1f974579654941ce460321533c1e9b81839003c177c3c397","observation_id":"e3b6e19c-f006-4094-a4f2-de465d4cb624","resolution":{"observed_at":"2026-08-12T13:46:02.177003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.154514Z","title":"Quasar-vit: Hardware-oriented quantization-aware architecture search for vision transformers,","venue":null,"work_id":"53e0675e-a2df-470a-b920-99d83809e37a","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.126214Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6c2681a82faabeb284488484f25097b8e9601cb169bd7f53fc97c91eae953ac4","observation_id":"fd869e77-bcc5-47ed-ae71-1f27620566bb","resolution":{"observed_at":"2026-08-12T13:46:02.160011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.139309Z","title":"High-performance fpga-based cnn accelerator with block-floating-point arithmetic,","venue":null,"work_id":"36111061-6492-4b60-9795-6d91e9b96060","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.130338Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:c718ffaaee2151612fc58928a3f87dc9b8a1abcc49a2406928490848611dbcca","observation_id":"995b9f46-e3c0-4229-aeed-91e1b6697a4a","resolution":{"observed_at":"2026-08-12T13:46:02.144687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.121523Z","title":"Awq: Activation-aware weight quan- tization for llm compression and acceleration,","venue":null,"work_id":"6c99f929-c8f6-4643-ad38-4e550530ce0b","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.134742Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:a41e1bb3041aa7040fc0d68593ac1d4a2568e15769ff81e91bedaee44c15d07a","observation_id":"ac37e816-1469-47bf-b6f8-e6b369beb443","resolution":{"observed_at":"2026-08-12T13:46:02.127154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04532","last_updated":"2025-05-01T02:14:05Z","snapshot_observed_at":"2026-08-13T00:12:52.272658Z","submitted_at":"2024-05-07T17:59:30Z","title":"QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04532","snapshot_observed_at":"2026-08-12T13:46:01.139346Z","title":"Qserve: W4a8kv4 quantization and system co-design for efficient llm serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.139346Z"},"links":{"cited_paper":"/paper/2405.04532","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:df93d0655d5c8f987424e15c31ae83c5c45e6a654c2eab9e87c058c6eb5220f3","observation_id":"0bd651d8-2194-4958-bb11-36374a362fd2","resolution":{"observed_at":"2026-08-12T13:46:01.139346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17888","last_updated":"2023-05-29T05:22:11Z","snapshot_observed_at":"2026-08-12T12:16:31.800541Z","submitted_at":"2023-05-29T05:22:11Z","title":"LLM-QAT: Data-Free Quantization Aware Training for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17888","snapshot_observed_at":"2026-08-12T13:46:01.145166Z","title":"Llm-qat: Data-free quantization aware training for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.145166Z"},"links":{"cited_paper":"/paper/2305.17888","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:2bb1aac11132b77c2eb274b535bbd05768532d662f9b721b519aeb2b718a0ab8","observation_id":"96581e0b-fb9d-4487-9eeb-24b973f4971b","resolution":{"observed_at":"2026-08-12T13:46:01.145166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.106782Z","title":"Kivi: A tuning-free asymmetric 2bit quantization for kv cache,","venue":null,"work_id":"a9e1dfd3-7694-48d8-bc7c-7fc607d28cfb","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.149988Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:b25b54f2b283bdfc76ef769ad54f612ccf91d6d8dbd7a902cfa05ee714a79869","observation_id":"938449d2-5a6d-4ae6-93af-c87b1417f81a","resolution":{"observed_at":"2026-08-12T13:46:02.111294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.089499Z","title":"Dis- tilling bit-level sparsity parallelism for general purpose deep learning acceleration,","venue":null,"work_id":"cb5ce676-8954-4496-a93f-66c0b58a50df","year":2021},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.154781Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e918a62514563b6c4dfebc2bb71a12ac51b4b96737cfd86b31afea6634acbf4a","observation_id":"9fac23e3-43b4-470b-9fe0-a571e64a12c3","resolution":{"observed_at":"2026-08-12T13:46:02.094997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.060174Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption,","venue":null,"work_id":"1436f428-8a20-4a8b-917a-fec16845daf5","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.159281Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:4d9872cb39c6b199f3e506d57e4641d821bd27612f7a9a8786715d781f62b33f","observation_id":"9f43329e-2dac-445d-80dc-259f8ef78496","resolution":{"observed_at":"2026-08-12T13:46:02.071054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17870","last_updated":"2024-10-18T02:01:18Z","snapshot_observed_at":"2026-08-12T22:36:54.098855Z","submitted_at":"2024-09-26T14:17:58Z","title":"Efficient Arbitrary Precision Acceleration for Large Language Models on GPU Tensor Cores","version":2},"cited_work":{"arxiv_id":"2409.17870","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.17870","snapshot_observed_at":"2026-08-12T13:46:01.503179Z","title":"Efficient Arbitrary Precision Acceleration for Large Language Models on GPU Tensor Cores","venue":"cs.LG","work_id":"e493c2d3-c1a8-4bc8-b65e-e65a24f82326","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.164716Z"},"links":{"cited_paper":"/paper/2409.17870","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:683044f84ed88e2c09f730856000bca24913c1d70b736b1db2787bfa3210e636","observation_id":"fda3f6a9-e9e4-4d15-b6ee-fed49585f9ee","resolution":{"observed_at":"2026-08-12T13:46:01.510626Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.040470Z","title":"Fpnew: An open-source multiformat floating-point unit architecture for energy-proportional transprecision computing,","venue":null,"work_id":"3e92e701-805e-4134-87df-809bdc38887c","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.169746Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:3495807dda366eaab64d06772c5c34098bb9570ef1f162be3351098b036e426c","observation_id":"92cb1726-fe3b-4af0-9975-a35027aa97d6","resolution":{"observed_at":"2026-08-12T13:46:02.047111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.023923Z","title":"The penn treebank: Anno- tating predicate argument structure,","venue":null,"work_id":"00e995ad-832f-4056-a84c-9946c25afdd8","year":1994},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.175099Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:5c76585965447dd25c7137d6954e86d04d7544a5a0f66a0bcf64cd3c785b58c3","observation_id":"ab2800b6-3577-4014-99cc-55e9691b7a41","resolution":{"observed_at":"2026-08-12T13:46:02.029380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:02.005235Z","title":"Pointer sentinel mix- ture models,","venue":null,"work_id":"062560e2-1a3f-42fa-94bf-cba761eed757","year":2017},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.180120Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:01a83919df489fc97fd700a6a4cdfaba5b9452cee9ede5ae95d213d4f91935cd","observation_id":"53655596-273c-437f-8332-4a6e8b23254c","resolution":{"observed_at":"2026-08-12T13:46:02.011649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.989325Z","title":"Flexblock: A flexible dnn training accelerator with multi-mode block floating point support,","venue":null,"work_id":"a6c16ab3-feeb-467e-add5-c97e14fc513c","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.184647Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:3e855f2b01864805dbe46f262b25dada5f02eb8007bc2f9cd0ec1dd512f4d210","observation_id":"c70b0fc3-7afe-4fca-9688-f206d80ec28d","resolution":{"observed_at":"2026-08-12T13:46:01.994256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.970911Z","title":"Cutlass,","venue":null,"work_id":"17a07bd0-aa4c-4f69-8b9c-e0c1c2504015","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.189062Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:56362850115ed8cea89bc41ebabbdade42774123032469f2c0c25747a730171e","observation_id":"42f71867-f06c-4e26-8f0f-df6241d81ce7","resolution":{"observed_at":"2026-08-12T13:46:01.977580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T13:46:01.193821Z","title":"Gpt- 4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.193821Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:b7dc4f9aadf984c17b2e816b57ef0ce9d0030c81f14fb56ea42ddbdebf228292","observation_id":"aaa2dfc7-d314-412d-b7be-4e114a599e64","resolution":{"observed_at":"2026-08-12T13:46:01.193821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.950306Z","title":"LUT-GEMM: Quantized matrix multipli- cation based on LUTs for efficient inference in large-scale generative language models,","venue":null,"work_id":"c6d48234-d0ff-4f62-bbd5-c970cb0c682a","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.198965Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9e60faa5c01f9c55c3424c3d6ab271875047c5bdb090bf78da86d07c7b15adf4","observation_id":"38f48fab-2824-488e-99e6-22a062854016","resolution":{"observed_at":"2026-08-12T13:46:01.956504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.930541Z","title":"Exploring the limits of transfer learning with a unified text-to-text transformer,","venue":null,"work_id":"11737760-2828-4127-b5d9-ea7fdf01907d","year":2020},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.203051Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:3f57bb5e97912fbdeaf27e5ac15d9067de6cc3ab03109f0b2b2c5613e221d59c","observation_id":"842a34fe-7a8a-445c-af99-adb9fc1a11ab","resolution":{"observed_at":"2026-08-12T13:46:01.937328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.913979Z","title":"Omniquant: Omnidirectionally calibrated quantization for large language models,","venue":null,"work_id":"f29737a1-cf8b-4dfa-ab17-d89d589a8ec8","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.207048Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:871b589fb56d165edf2c611b2fa994dc0b44ef9f9144609e42d8ea27467ab0ee","observation_id":"a2e29963-82d3-4df9-bd66-09724eabba27","resolution":{"observed_at":"2026-08-12T13:46:01.918778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.896755Z","title":"Bit fusion: Bit-level dynamically composable architecture for accelerating deep neural network,","venue":null,"work_id":"c09bd469-c3ea-4c65-a58a-02413c431444","year":2018},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.211499Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:614f1e2e38635643f219e1b0d797f25069ea7b76eef643c4bd02068a4d338cfb","observation_id":"0c831a98-4474-4151-be7d-780a15bb36ad","resolution":{"observed_at":"2026-08-12T13:46:01.902106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.877610Z","title":"Bitwave: Exploiting column-based bit-level sparsity for deep learning accelera- tion,","venue":null,"work_id":"bae62f10-11a2-41cd-bb48-8b06b64d5fef","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.215760Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:d9fa1a402861e58561939f723d5401dced45572515a48684b267101ba8060cb1","observation_id":"6fc9dfc1-34c9-4723-bbb3-a58c01237086","resolution":{"observed_at":"2026-08-12T13:46:01.883768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.861263Z","title":"Dissecting tensor cores via microbenchmarks: Latency, throughput and numeric behaviors,","venue":null,"work_id":"39c31d5f-f5ae-4dfb-86ae-b3105c5a9a38","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.219759Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:9928050fc72a0bc32c8dd4b6279975898043bcde2ccf4503d329e0a4cd46085f","observation_id":"29c5eea0-0148-490a-bccc-b7a1fa42f4ca","resolution":{"observed_at":"2026-08-12T13:46:01.866350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-12T13:46:01.224872Z","title":"Gemma: Open models based on gemini research and technology,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.224872Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:160467b1fe275c3e299763c5b184de11a31911195a11a299c9f4464aa4c954d8","observation_id":"9f38003d-2c2e-46f5-a22d-73d7d17f5816","resolution":{"observed_at":"2026-08-12T13:46:01.224872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.842387Z","title":"Bebert: Efficient and robust binary ensemble bert,","venue":null,"work_id":"c597c36f-1992-4d40-b9d9-da5617f2ec8c","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.230172Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:d2929721ed02f7eaf451a52ae0b5a9de93cd9652e20580508f3e7c88fe3eb11a","observation_id":"141d78a5-df15-482d-b957-e9c47fd67d6a","resolution":{"observed_at":"2026-08-12T13:46:01.847626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-12T13:46:01.235170Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.235170Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:cd22e7f537b95c80831a9445c70792eda247fa29592d097c00a4061778e84f6f","observation_id":"11d12725-bcca-489b-a2a9-cfa99513dfc3","resolution":{"observed_at":"2026-08-12T13:46:01.235170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-12T13:46:01.241211Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.241211Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:a1973dbafb42ae8c7cccaec61d45873db78bfef5be9c3029bf712b945036bc7a","observation_id":"5c8e640e-b043-4082-8752-c673a4b40678","resolution":{"observed_at":"2026-08-12T13:46:01.241211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04396","last_updated":"2024-06-04T04:51:52Z","snapshot_observed_at":"2026-08-13T04:24:47.509375Z","submitted_at":"2024-02-06T20:52:12Z","title":"QuIP#: Even Better LLM Quantization with Hadamard Incoherence and Lattice Codebooks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04396","snapshot_observed_at":"2026-08-12T13:46:01.246476Z","title":"Quip#: Even better llm quantization with hadamard incoherence and lattice codebooks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.246476Z"},"links":{"cited_paper":"/paper/2402.04396","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6806dbd0834e914bb5f0d0fa38a5c7b33efed74a9e88e0c018bc7f1efe9a06d6","observation_id":"46eab912-d74a-47a8-b1b4-90ff15b5743c","resolution":{"observed_at":"2026-08-12T13:46:01.246476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.824674Z","title":"Bsvit: A bit-serial vision transformer accelerator exploiting dynamic patch and weight bit-group quantization,","venue":null,"work_id":"875a4acd-0c89-418c-9aad-2f89d93ee3c3","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.252804Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:660c4e93645054508fe30443a6377f90b9ca8378a631c901ae75153c8ba3eac2","observation_id":"8714c3f3-58f9-4e62-b876-960241b8ce18","resolution":{"observed_at":"2026-08-12T13:46:01.830986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.808713Z","title":"Haq: Hardware-aware automated quantization with mixed precision,","venue":null,"work_id":"50dfaac0-e71d-4482-9178-83f6908e95c9","year":2019},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.257602Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6abdcc3ed06695f90ad631182b169e27180506e8b7805ad47c0df0adeac01547","observation_id":"7117e91a-1dc4-4735-8b60-08834df4a5c0","resolution":{"observed_at":"2026-08-12T13:46:01.813986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.790812Z","title":"Outlier suppression: Pushing the limit of low-bit transformer language models,","venue":null,"work_id":"fe61d7a9-23c5-4458-b6b7-f032a542d2ed","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.263418Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:91caef0b733a958fede27b9d96286373f59038a4a745a3cb2536f925526b1182","observation_id":"7357fcfa-01dc-47ce-984c-24f3c5b8c6d7","resolution":{"observed_at":"2026-08-12T13:46:01.796440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.774302Z","title":"Quant-llm: Accelerating the serving of large language models via fp6- centric algorithm-system co-design on modern gpus,","venue":null,"work_id":"4829e6c8-09a7-47b1-b8b4-c85af7d960de","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.268642Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:f41bc02170795a209479e681d3a09e60983ca4f2a8d16e4d27424e914cc9b87b","observation_id":"e45a072e-fb67-41d6-b90b-9c748cf3f417","resolution":{"observed_at":"2026-08-12T13:46:01.780173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.757965Z","title":"Smoothquant: Accurate and efficient post-training quantization for large language models,","venue":null,"work_id":"3081fba4-1f44-48bb-9f73-cc65dc4cb8ef","year":2023},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.274153Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:3bb75e1b2c66850d48084a43b792363202e507832d589aa426a5e373fd7e4f02","observation_id":"a71eaf87-6323-4f73-9d05-19aca2ce0d6b","resolution":{"observed_at":"2026-08-12T13:46:01.762681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.739649Z","title":"Efficient streaming language models with attention sinks,","venue":null,"work_id":"6aa653da-9fc5-427f-8071-87b76c356662","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.279979Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:f602da19ce6fb1b91572ac716ef157c37df28109419efc4bb172400455cb54d7","observation_id":"bcbc3551-c787-4ab8-a8e0-5a7133c1a315","resolution":{"observed_at":"2026-08-12T13:46:01.745057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11295","last_updated":"2024-11-29T11:47:55Z","snapshot_observed_at":"2026-08-13T04:17:01.673281Z","submitted_at":"2024-02-17T14:26:57Z","title":"OneBit: Towards Extremely Low-bit Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11295","snapshot_observed_at":"2026-08-12T13:46:01.284612Z","title":"Onebit: Towards extremely low-bit large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.284612Z"},"links":{"cited_paper":"/paper/2402.11295","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:198f01a4ac53494ee7fb99e54eac85ed44a748237a5aabd7710b821b9f93861d","observation_id":"40732215-a09e-41ab-9da8-fce2c45ab70d","resolution":{"observed_at":"2026-08-12T13:46:01.284612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.719478Z","title":"Kv cache compression, but what must we give in return? a comprehensive benchmark of long context capable approaches,","venue":null,"work_id":"65b30caa-15f5-4833-bb89-9bb89e6af20a","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.289754Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:ee327e4e8375a8fbc1ccf47736854af0d135f050de762f81dad7d8d35ee02668","observation_id":"486b959b-41fa-45b4-854b-7615f86f023f","resolution":{"observed_at":"2026-08-12T13:46:01.726050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16363","last_updated":"2024-05-01T20:42:28Z","snapshot_observed_at":"2026-08-13T04:10:18.599255Z","submitted_at":"2024-02-26T07:33:05Z","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16363","snapshot_observed_at":"2026-08-12T13:46:01.294427Z","title":"Llm inference unveiled: Survey and roofline model insights,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.294427Z"},"links":{"cited_paper":"/paper/2402.16363","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:11eee84523ce382629d6ffe462bd694e70a0f4abf676bc4302050732eb62c4e0","observation_id":"ff4d496e-4129-4775-9c74-ebbb851dd57d","resolution":{"observed_at":"2026-08-12T13:46:01.294427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.697876Z","title":"Mokey: Enabling narrow fixed-point inference for out-of-the-box floating-point transformer models,","venue":null,"work_id":"9aaf271f-e395-4b17-8d5e-bcaf4867339e","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.299569Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:98f61bbb487a5129abc5f0f278d7e2028d99547f7d14ad3a53a39ae9c6122bd1","observation_id":"85aeb8e5-3c60-483a-87ce-7b2378c2c551","resolution":{"observed_at":"2026-08-12T13:46:01.705368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.676573Z","title":"Fast: Dnn training under variable precision block floating point with stochastic rounding,","venue":null,"work_id":"75fc523d-e5d1-4bd3-a05b-d51abfba1272","year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.303655Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6b6c791f12fa1aebfa0d747b48fd2a9613aac49919cdf53276dedb45b7ffc538","observation_id":"aca341d1-8958-4999-ade4-1e25ceec5f95","resolution":{"observed_at":"2026-08-12T13:46:01.682402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-12T13:46:01.308601Z","title":"Opt: Open pre-trained transformer language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.308601Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:1255b67a2907a3e1839f08337146acd65e4e8dab3688f781064dbda10140466b","observation_id":"50f869ac-ab0d-42bf-85f6-51a7c069feca","resolution":{"observed_at":"2026-08-12T13:46:01.308601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.653262Z","title":"Cam: Cache merging for memory-efficient llms inference,","venue":null,"work_id":"b3fdf0ba-d5a1-468a-8cd1-15afec1d0bcd","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.313207Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6767e9fde40108a3548f07b3769ed1654ea3fbb82d5777f6d937ab022593a7cb","observation_id":"60706211-3401-4a13-b602-09912f9bbf66","resolution":{"observed_at":"2026-08-12T13:46:01.659492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.632187Z","title":"H2o: Heavy-hitter oracle for efficient generative inference of large language models,","venue":null,"work_id":"44a41251-74c6-48b7-9f2a-ad6c6699dba7","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.317190Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:e77ee620298fdc40fa0183df1229a7a16855e3dd87e90cbbd006b25368a52b0d","observation_id":"a419e5d6-a114-4446-9c43-703baa852076","resolution":{"observed_at":"2026-08-12T13:46:01.639814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:46:01.614579Z","title":"Atom: Low-bit quantization for efficient and accurate llm serving,","venue":null,"work_id":"49b53d44-c5ea-4c3a-9848-6b2ec8757aa0","year":2024},"citing_paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-12T13:46:01.321581Z"},"links":{"citing_paper":"/paper/2411.15982"},"observation_digest":"sha256:6b12db78a71b2b680acbbf0671f566b74157db70d3b891640621c0f6c93fe005","observation_id":"5e21a48c-cfbe-473b-a856-68a0bb2292e9","resolution":{"observed_at":"2026-08-12T13:46:01.621468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.15982","last_updated":"2024-11-24T20:59:39Z","latest_version":1,"primary_category":"cs.AR","snapshot_observed_at":"2026-08-12T13:37:48.048787Z","submitted_at":"2024-11-24T20:59:39Z","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":1,"verified_fuzzy":65},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 0 inbound Pith citation observations for arXiv:2411.15982."}