{"as_of":"2026-08-21T16:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:da91795b4cb7f5a3e035be02e0d850ccde1ff75c61462c9904268340a6fa669e","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-12T06:20:07.112455Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.02893/citation-record","integrity":"/paper/2607.02893/integrity","json":"/paper/2607.02893/citation-record.json","paper":"/paper/2607.02893"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Introducing apple’s on-device and server foundation models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:473de9b040994efa6c642e950fd21f947d20c0f80b6da76d3c35e4895d072d6d","observation_id":"9c430808-cf45-4e16-8007-892318372983","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:9ccaaea45dd92af5a0953a777c48f5f564c01dc3861826abc2e79f846927d321","observation_id":"7e085793-e60d-4b85-8ee5-cebd0e1123dc","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1308.3432","last_updated":"2013-08-15T15:19:34Z","snapshot_observed_at":"2026-08-14T04:51:04.817737Z","submitted_at":"2013-08-15T15:19:34Z","title":"Estimating or Propagating Gradients Through Stochastic Neurons for Conditional Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1308.3432","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Estimating or propagating gradients through stochastic neurons for conditional computation.arXiv preprint arXiv:1308.3432, 2013","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/1308.3432","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:66e9aabe3810eac2ae429c83f8accede6afd8383d1afa75c2bb63b28ece93759","observation_id":"abb8d770-be34-4ed0-a0c8-61a31fe30368","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Rethinking differentiable search for mixed-precision neural networks","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:a2bf6aee352b1ecabc9741edfdfcc700ae04098cf4b936865fb9d64d5fb768ca","observation_id":"949dd14f-7af8-4634-8afc-ee5d7a3cb541","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06722","last_updated":"2025-08-06T22:52:22Z","snapshot_observed_at":"2026-08-18T04:14:35.157203Z","submitted_at":"2024-10-09T09:45:01Z","title":"Scaling Laws For Mixed Quantization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06722","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2410.06722","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:4f7a4905b9b1f203bf6a639e4809a8bc761b20be123ffb955abad0360f203e2c","observation_id":"c9d03994-6431-4cee-842a-b956f5d8db66","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18556","last_updated":"2026-05-15T09:34:30Z","snapshot_observed_at":"2026-08-13T18:41:24.702288Z","submitted_at":"2026-04-20T17:45:47Z","title":"GSQ: Highly-Accurate Low-Precision Scalar Quantization for LLMs via Gumbel-Softmax Sampling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.18556","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"GSQ: Highly-accurate low-precision scalar quantization for LLMs via gumbel-softmax sampling","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2604.18556","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:eea4d51cfb63f08f508a2cef0e046b5d04ac0f486ca1969cf252b77ccc3ffd31","observation_id":"801f5d72-5310-41b7-b251-867a73c40618","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"The case for 4-bit precision: k-bit inference scaling laws","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:437bf79631a5b2a064ba8f4335c309f0a8d49cca0aa2bc249d65b5b96f849281","observation_id":"af44db72-7cb6-4a08-a63b-4177888d8a42","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"LLM.int8(): 8-bit matrix multiplication for transformers at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:01d694967d9570e6e77d585c7bed92f7dfac58f8a9d59ed2938696a6c153babf","observation_id":"9e728a3a-de8d-4aca-b012-00ae616d276d","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Mahoney, and Kurt Keutzer","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:58eac7ce5266ebc126d01498f87c54eb38adc7b40f07b97f8193fb5f2b739652","observation_id":"8558b211-2862-4c5c-973d-8ae8e7e7faae","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.09141","last_updated":"2024-07-12T10:19:02Z","snapshot_observed_at":"2026-08-16T13:34:48.836282Z","submitted_at":"2024-07-12T10:19:02Z","title":"Accuracy is Not All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.09141","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Accuracy is not all you need","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2407.09141","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:7e7f236341d6ec6364459951b6a5a567a5d31574811ce570065a34ca68069037","observation_id":"9d004b53-0b49-480b-ab1b-681f1022046d","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Extreme compression of large language models via additive quantization.International Conference on Machine Learning (ICML), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:328bd80d0e06cd930d606efe8189b2371eccd6c468c3b398d0e5fd25756f218a","observation_id":"e288edd0-1a2f-4c34-8485-0b42eabb054b","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07759","last_updated":"2023-05-24T23:30:43Z","snapshot_observed_at":"2026-08-20T14:47:29.320021Z","submitted_at":"2023-05-12T20:56:48Z","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07759","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"TinyStories: How small can language models be and still speak coherent english?arXiv preprint arXiv:2305.07759, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2305.07759","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:6b6a1373b647a20f9cf7e636649099e7418f118aba7e495d4c8ddd0094f1740c","observation_id":"aaf3d5c4-31e7-4dfe-ad78-39b92fa53efc","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.04115","last_updated":"2026-07-14T14:04:32Z","snapshot_observed_at":"2026-08-02T05:34:23.902031Z","submitted_at":"2026-06-02T18:23:20Z","title":"dMX: Differentiable Mixed-Precision Assignment for Low-Precision Floating-Point Formats","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.04115","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"dMX: Differentiable mixed-precision assignment for low-precision floating-point formats.arXiv preprint arXiv:2606.04115, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2606.04115","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:cc01eddea740213ee60aec58f45194f90d8c3952652fdb90ed8e12f834ca842b","observation_id":"f96a24be-7752-4512-886e-858bc8952ebf","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"GPTQ: Accurate post-training quantization for generative pre-trained transformers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:f801b7bbe082530c5b0e8ee723f643c1a85e751341b608f789266aa1e3f5c291","observation_id":"15a262b0-6465-4ab1-be80-6e39b3155bd0","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-08-16T18:00:58.008096Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Distilling the knowledge in a neural network.arXiv preprint arXiv:1503.02531, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:76195abf69a1e28c3746d8e38e5c28e445f5426d4fc01da6d00e504c8dcbf49a","observation_id":"0fa2381c-37b7-4ac1-af0d-4609dbd2a7a5","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14917","last_updated":"2025-05-25T08:58:37Z","snapshot_observed_at":"2026-08-18T22:14:24.661825Z","submitted_at":"2024-05-23T16:21:48Z","title":"SliM-LLM: Salience-Driven Mixed-Precision Quantization for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14917","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"SliM-LLM: Salience-driven mixed-precision quantization for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2405.14917","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:da204ba1f49c8f6d2c484ecd8501ab8ade280d547509161cd7ce859e18ef3c21","observation_id":"269cae30-78e8-46b2-963d-9db092a59491","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Categorical reparameterization with Gumbel-Softmax","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:2f9df910039728f83565646bf117559cd68021541bbaf1547fc6a8bbc8580b13","observation_id":"ad9a1d15-c8f4-41b3-9258-40ccfbe98fbf","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Spector, Blake Bordelon, Niklas Muennighoff, Mansheej Paul, Cengiz Pehlevan, Christopher Ré, and Aditi Raghunathan","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:a3cdf5faae765bb79918f4160f4bb7efa232a939ffb6464c95db2065b8f52af8","observation_id":"992beea6-92ca-4667-9ac0-11d039a87158","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"AWQ: Activation-aware weight quantization for on-device LLM compression and acceleration","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:5186c654fe80a2bec4c40aae92e0731ba9f359189ee14353edab33b697441e59","observation_id":"9b2591f5-5c86-4cd8-af98-aafadbeb9c20","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"DARTS: Differentiable architecture search","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:11dd9693ad9ae95c94236dde75c641b6d77a0879eee6f68083ccfe6146065b57","observation_id":"06c75d7d-8003-4f87-83b2-2d7f07095458","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"ParetoQ: Improving scaling laws in extremely low-bit LLM quantization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:829405436e491066e578ab08fa6b906a1982401af38b024370f4a96c52378f0a","observation_id":"f14718cf-a609-4c94-8c2b-6104313d3f42","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17764","last_updated":"2024-02-27T18:56:19Z","snapshot_observed_at":"2026-08-14T15:17:37.878305Z","submitted_at":"2024-02-27T18:56:19Z","title":"The Era of 1-bit LLMs: All Large Language Models are in 1.58 Bits","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17764","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"The era of 1-bit LLMs: All large language models are in 1.58 bits.arXiv preprint arXiv:2402.17764, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2402.17764","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:f861f7b5e2dd4bd95eaa549b3d5f5b8cf48ee062096f1f6606fdb7ae8ad67e8b","observation_id":"e0b92520-541d-4826-998f-2acfe0ed7675","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Maddison, Andriy Mnih, and Yee Whye Teh","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:4862aef78c8cdfbf89a733db35538670c6331b0772b1004d0d2094123540e144","observation_id":"5af0ec40-ac71-4445-aeca-0d80212e9a20","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Introducing NVFP4 for efficient and accurate low-precision in- ference","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:7a30d16cf44b9677ad93ff75a5b65708fbd0eeb92970bc5a0f9d2842acb0d204","observation_id":"64b1651d-ea0f-4cd4-8300-a369da0b205f","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Pretraining large language models with NVFP4","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:1073578d35d45dd87b67cfe526790175720a5a1fa45b8d8ed1079ed70cd1c453","observation_id":"83e12407-03ff-47af-aee0-6e8fa9066c6a","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17557","last_updated":"2024-10-31T11:37:49Z","snapshot_observed_at":"2026-08-20T16:41:47.138179Z","submitted_at":"2024-06-25T13:50:56Z","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17557","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"The FineWeb datasets: Decanting the web for the finest text data at scale","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2406.17557","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:e9b3aea661cad17066043078301d107a401f5fd60a7fd2ab6feaedfd76557287","observation_id":"14d41e4d-5a88-4599-ac44-8f14fefca371","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Bonsai: 1-bit and ternary (1.58-bit) language models for on-device inference.https: //prismml.com/news/ternary-bonsai, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:d93ac93f25c2d62b7a5ae9624e85d835605493ef8ab67e4bf6414a0b004b0ea4","observation_id":"e19aea9f-d1c4-4e83-8a55-720c0c0930b5","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Qwen3 technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:1da01069d74bcad4d30462c532ba8ce9e962b1aaeabe5c15a661750c3d54e82f","observation_id":"afe9501a-4b5c-41c2-8cf1-c9eafe0cc659","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Language models are unsupervised multitask learners.OpenAI Technical Report, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:9702a25e2c8fd8a16be655870fb581e43215ea9d672f852d5a123b9eede455bd","observation_id":"868adaf7-cb4b-41de-9257-d11f7009c54c","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Neural hashing: The future of search","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:5e4fdb81157a5042da84c5a01ed0c0c7bedd9f46f8517784f0528a803bae893d","observation_id":"981a02bd-6b5d-4975-aadb-7255857c6682","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-08-11T06:21:56.129166Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"GLU variants improve transformer.arXiv preprint arXiv:2002.05202, 2020","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:c9c666b6090df8635bb9cee1fd8bcbd8ad36e3bc7d8a12c978cbe067ea5f8dc9","observation_id":"e283b8ec-65f9-449b-b47d-b87b96e48e7c","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"RoFormer: Enhanced transformer with rotary position embedding.Neurocomputing, 568, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:6b8d060f837be5c7640861f0decc18832037ecd7ab2abbee8925ca86596a7879","observation_id":"a24a10cb-9ae6-49a7-849f-b5162bde7830","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"QuIP#: Even better LLM quantization with hadamard incoherence and lattice codebooks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:ad7f93833c4a9df2515fd81772cd5a1da2e4fd7cb134ecbd7d0aa460b6e73567","observation_id":"cc47bd7f-209c-4fdb-872c-96e49ec5373e","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Unsloth dynamic GGUF quants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:c171478dd7565df20eba9ae3a2e2c9d381d50d4b9b87592a331da6413114032a","observation_id":"366a9eea-fd33-4558-9870-fb271f324b47","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11453","last_updated":"2023-10-17T17:59:15Z","snapshot_observed_at":"2026-08-15T04:43:14.762539Z","submitted_at":"2023-10-17T17:59:15Z","title":"BitNet: Scaling 1-bit Transformers for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11453","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"BitNet: Scaling 1-bit transformers for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2310.11453","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:b5861213e9f264b94b77267ce02a31d9a9c7f035ac192506e07dcda62ef1f591","observation_id":"a938dd76-9e7a-4734-8305-a0b0382df0da","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"HAQ: Hardware-aware automated quantization with mixed precision","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:fb57a044c3613e11ad6cc0d7c4a9318d84b2d59923d43ad546ed1002fefc0365","observation_id":"dfb8442f-00c8-45aa-9e58-f01af8a88415","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1812.00090","last_updated":"2018-11-30T23:15:45Z","snapshot_observed_at":"2026-08-14T17:50:18.985454Z","submitted_at":"2018-11-30T23:15:45Z","title":"Mixed Precision Quantization of ConvNets via Differentiable Neural Architecture Search","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1812.00090","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"Mixed precision quantization of ConvNets via differentiable neural architecture search.arXiv preprint arXiv:1812.00090, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/1812.00090","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:a16588dba87a5f632f16f3b6df29dab98eb0def2f8056e47cd4e78a76e0326eb","observation_id":"f6848eb2-332b-4dc4-ad76-1a9d9bd7444b","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:e4a52e017ae7e4bc027aaebe60ca74478b4e8f29b6da741bdccf2bad5b8c8e8a","observation_id":"bd88fe1c-a4f4-46c8-8f7e-f1268e9fd667","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14590","last_updated":"2026-04-22T15:43:48Z","snapshot_observed_at":"2026-08-16T01:35:40.607992Z","submitted_at":"2024-12-19T07:15:15Z","title":"MixLLM: LLM Quantization with Global Mixed-precision between Output-features and Highly-efficient System Design","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14590","snapshot_observed_at":"2026-07-12T06:20:07.112455Z","title":"MixLLM: LLM quantization with global mixed- precision between output-features and highly-efficient system design","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-12T06:20:07.112455Z"},"links":{"cited_paper":"/paper/2412.14590","citing_paper":"/paper/2607.02893"},"observation_digest":"sha256:af3999c9cc619985015b2fba1136116eeec409ea082cc740c4bfb926ddb33c03","observation_id":"58e27d8e-bafb-43ad-9522-5e1bd0616350","resolution":{"observed_at":"2026-07-12T06:20:07.112455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.02893","last_updated":"2026-07-03T02:42:04Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T08:01:33.845481Z","submitted_at":"2026-07-03T02:42:04Z","title":"Variable Bit-width Quantization: Learning Per-Group Precision for \"Bigger-but-Smaller\" Language Models"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 0 inbound Pith citation observations for arXiv:2607.02893."}