{"as_of":"2026-08-13T03:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:21b026935f0391c8852ab1e90407fcdc3bd95a35682ade9005dc59a271377698","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:56:39.808556Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T01:08:14.247342Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T08:26:24.739905Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"cited_work":{"arxiv_id":"2412.12639","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.12639","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Falcon: Faster and parallel inference of large language models through enhanced semi-autoregressive drafting and custom-designed decoding tree","venue":null,"work_id":"66667291-fcf4-46da-a9da-821b1f13753d","year":2024},"citing_paper":{"arxiv_id":"2605.08632","last_updated":"2026-05-09T02:50:58Z","snapshot_observed_at":"2026-08-12T22:50:31.398809Z","submitted_at":"2026-05-09T02:50:58Z","title":"PARD-2: Target-Aligned Parallel Draft Model for Dual-Mode Speculative Decoding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T01:08:14.247342Z"},"links":{"cited_paper":"/paper/2412.12639","citing_paper":"/paper/2605.08632"},"observation_digest":"sha256:7d87e2e02792332d8e0cefcfe808b55c93202cea99475f3adb2b1304b3eb1cd1","observation_id":"4e8c7bb5-17af-47b6-a400-6099c7817c98","resolution":{"observed_at":"2026-05-12T08:26:24.741992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.12639/citation-record","integrity":"/paper/2412.12639/integrity","json":"/paper/2412.12639/citation-record.json","paper":"/paper/2412.12639"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.544245Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.544245Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:6bf4fe24322b339d2be47e2cbdba01eea822bf1bf13d399c6a2c7fc43d860c6e","observation_id":"088995a7-78ae-4a33-96d8-086b3ba695a1","resolution":{"observed_at":"2026-08-11T13:56:39.544245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.549623Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.549623Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:e15db51a7ff3bcc4c6b7712cec66391aceba7c6b2b67abc742e793f676ae3701","observation_id":"1095de3a-02e5-48b5-8e39-cdeb02d13c4d","resolution":{"observed_at":"2026-08-11T13:56:39.549623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.556071Z","title":null,"venue":null,"work_id":"beeac847-b0e6-4632-a5c2-8743561cd189","year":2022},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.555810Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:01f621757bc5047a835bdc6ab11e7dc0906b72f21d86cfaeed7db5eeec31bb83","observation_id":"cd21b300-3a5a-4f28-bccd-710ef8584d9c","resolution":{"observed_at":"2026-08-11T13:56:40.560090Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-07-06T17:17:56.276857Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-11T13:56:39.561574Z","title":"D.; Chen, D.; and Dao, T","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.561574Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:f6feddc9411a639b787f9cf870095e1efc3342a502156008e2e88985d346f3ef","observation_id":"1f4542ba-b36b-4454-b281-de72137be92f","resolution":{"observed_at":"2026-08-11T13:56:39.561574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-08-10T21:52:49.568983Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-11T13:56:39.568070Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.568070Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:510555064d3d58e370ab3bd0eb4710ab355de9623b81b58ddb9bb570266a913a","observation_id":"c8fa522f-0984-42ad-99c2-8f573c815683","resolution":{"observed_at":"2026-08-11T13:56:39.568070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-11T13:56:39.573168Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.573168Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:7a1b56a37b6aab4dab114056464cd769e5e4eef4b01f8874ce710054e6d7dcf0","observation_id":"b102a94c-ac3e-486c-b195-872c3f3f1a09","resolution":{"observed_at":"2026-08-11T13:56:39.573168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11462","last_updated":"2025-07-13T19:45:40Z","snapshot_observed_at":"2026-07-06T17:04:46.288600Z","submitted_at":"2023-12-18T18:59:46Z","title":"Cascade Speculative Drafting for Even Faster LLM Inference","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11462","snapshot_observed_at":"2026-08-11T13:56:39.580987Z","title":"C.-C.; and Huang, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.580987Z"},"links":{"cited_paper":"/paper/2312.11462","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:1e94681741022bd44ebe5dfa82609bc0605bc16532d18c106607a28d3c3cab74","observation_id":"85d3b22d-8c13-4fad-9bc1-0a8201207dca","resolution":{"observed_at":"2026-08-11T13:56:39.580987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T13:56:39.586758Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.586758Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d5e26e56fe203dd32c35b00ef99bf1c335a0f021a9e8a8fafb4acbf68cdfcded","observation_id":"49babc3b-b03a-4f88-9090-b3237ccf8188","resolution":{"observed_at":"2026-08-11T13:56:39.586758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.542624Z","title":"F.; Tao, D.; and Tu, Z","venue":null,"work_id":"31959b18-bbc0-4b2c-95b5-0d2fc9997b67","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.592117Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:63f24c4bba28e57c7892bfe3d3fe59aabb369e9f6a1e862b24585063885cf487","observation_id":"330b6d1d-7df0-4e21-8124-61eeafdf9cd6","resolution":{"observed_at":"2026-08-11T13:56:40.547022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.10360","last_updated":"2022-03-17T11:49:55Z","snapshot_observed_at":"2026-07-06T10:51:12.871009Z","submitted_at":"2021-03-18T16:30:26Z","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.10360","snapshot_observed_at":"2026-08-11T13:56:39.596093Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.596093Z"},"links":{"cited_paper":"/paper/2103.10360","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:2785fdaa4ed016b275236629472d79d13207e3d103df9a30557c32b62149cb0d","observation_id":"dfc0bfd4-2231-4a0d-b1f1-898a3cd10303","resolution":{"observed_at":"2026-08-11T13:56:39.596093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.530666Z","title":null,"venue":null,"work_id":"5c2f64cf-bc89-4343-b868-c1317585f30a","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.600341Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:008694ec18b52063be8bb31c0dba57f86c2e408e9de348855f7fc2dc2efa1380","observation_id":"2ada1f50-1235-4c8d-a9bc-f04c24a8fe48","resolution":{"observed_at":"2026-08-11T13:56:40.534753Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.515082Z","title":null,"venue":null,"work_id":"4a87edca-6fa3-489e-8f84-2ff6ee31a37b","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.603874Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:f4b78d38bdcd0c41450a6f10cb12b7a3268c2b0370742598f589f2d050463efe","observation_id":"6f84257a-97a4-414e-9fef-66e472a28cbf","resolution":{"observed_at":"2026-08-11T13:56:40.519615Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19737","last_updated":"2024-04-30T17:33:57Z","snapshot_observed_at":"2026-08-12T13:06:04.804423Z","submitted_at":"2024-04-30T17:33:57Z","title":"Better & Faster Large Language Models via Multi-token Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19737","snapshot_observed_at":"2026-08-11T13:56:39.607525Z","title":"Y.; Rozière, B.; Lopez-Paz, D.; and Synnaeve, G","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.607525Z"},"links":{"cited_paper":"/paper/2404.19737","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:fdc0d5f0cd6fdb8d5b81c20f6affc5d91531bb77ce6774ab93f43b5746464b72","observation_id":"175d18a3-778f-45f0-a3f4-316ff51a0d06","resolution":{"observed_at":"2026-08-11T13:56:39.607525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.612401Z","title":null,"venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.612401Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:527196431fcf4c48b7407027bf9afe7d2440abb85551295e3f6f611d67a581ee","observation_id":"af66018e-a323-43ff-a041-e8aa6ef34276","resolution":{"observed_at":"2026-08-11T13:56:39.612401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.489084Z","title":null,"venue":null,"work_id":"277bb584-0ea1-40a5-8e8d-16652e72540b","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.616664Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d91cb26d6e2e8ec908fe56d5b9bac0d59923f7e5abe39610217009fd28de1ab3","observation_id":"8ab821b5-56ec-4d84-af51-00a24bbfdb33","resolution":{"observed_at":"2026-08-11T13:56:40.493299Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.08717","last_updated":"2019-11-21T09:43:45Z","snapshot_observed_at":"2026-08-03T00:59:52.298231Z","submitted_at":"2019-11-20T05:48:31Z","title":"Fine-Tuning by Curriculum Learning for Non-Autoregressive Neural Machine Translation","version":2},"cited_work":{"arxiv_id":"1911.08717","doi":null,"metadata_source":"pith","pith_arxiv_id":"1911.08717","snapshot_observed_at":"2026-08-11T13:56:40.080049Z","title":"Fine-Tuning by Curriculum Learning for Non-Autoregressive Neural Machine Translation","venue":"cs.LG","work_id":"2ca4dab0-c145-48d7-9e98-5fa577606994","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.621807Z"},"links":{"cited_paper":"/paper/1911.08717","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:28a5c8f80182f469399cb91fe8979b66ca25874c112ad96ccdcad5c4abd84fef","observation_id":"8fa911fd-5676-4554-b832-7d0412bbc4be","resolution":{"observed_at":"2026-08-11T13:56:40.085206Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.472194Z","title":null,"venue":null,"work_id":"22100308-796c-4670-9e53-8426ea458e3b","year":2020},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.627068Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:0ea1470528ceb0acbcccf548c54c31334bf18e83a1853757c65aca12d71099e2","observation_id":"a288e5bd-b530-420d-939f-c51e55f907c5","resolution":{"observed_at":"2026-08-11T13:56:40.477145Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.11640","last_updated":"2021-12-22T03:06:27Z","snapshot_observed_at":"2026-08-06T13:41:43.963110Z","submitted_at":"2021-12-22T03:06:27Z","title":"Self-Distillation Mixup Training for Non-autoregressive Neural Machine Translation","version":1},"cited_work":{"arxiv_id":"2112.11640","doi":null,"metadata_source":"pith","pith_arxiv_id":"2112.11640","snapshot_observed_at":"2026-08-11T13:56:40.054637Z","title":"Self-Distillation Mixup Training for Non-autoregressive Neural Machine Translation","venue":"cs.CL","work_id":"44224712-0055-430e-9212-0aa052d1d177","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.631862Z"},"links":{"cited_paper":"/paper/2112.11640","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:7f884f02f006ce2ee53fcbb886ab54e33ed10690e9a8974dd149d95f30f10137","observation_id":"4482b961-7191-436d-a1f9-4a51fd5f0c59","resolution":{"observed_at":"2026-08-11T13:56:40.061510Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-11T13:56:39.638681Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.638681Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:ae0e7a490905e900faa754b1d66fa47287f81e821accac991f1672c6bc724bf6","observation_id":"e903e635-f4ba-40a0-a7b8-19bfd2aff34f","resolution":{"observed_at":"2026-08-11T13:56:39.638681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.642790Z","title":null,"venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.642790Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:71edc7be7a16bbb0af87bfc26bacae7f0a3e7ef49191ea7a82d3d1b35ffb54d8","observation_id":"47ab6896-f1db-4b66-b3b6-ac9762015293","resolution":{"observed_at":"2026-08-11T13:56:39.642790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.439155Z","title":"W.; Gholami, A.; and Keutzer, K","venue":null,"work_id":"0e439127-7e01-48e4-8675-3f5547b09844","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.647897Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:123eb1334d2a430f96f06969f8c514b4b43a9d507711f4b034c289871e661a0b","observation_id":"176e4f2f-8909-4647-a87e-36107980e473","resolution":{"observed_at":"2026-08-11T13:56:40.446329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.17192","last_updated":"2023-05-18T20:28:20Z","snapshot_observed_at":"2026-08-08T11:03:33.403601Z","submitted_at":"2022-11-30T17:33:28Z","title":"Fast Inference from Transformers via Speculative Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.17192","snapshot_observed_at":"2026-08-11T13:56:39.653053Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.653053Z"},"links":{"cited_paper":"/paper/2211.17192","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:0de9ba3dcb78e987e5320d8f5798baf815502b1de844e87f08586f9cb281d068","observation_id":"ba1f2e9a-4969-45f3-9bfd-3d67295ac2cc","resolution":{"observed_at":"2026-08-11T13:56:39.653053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15077","last_updated":"2025-03-04T13:58:39Z","snapshot_observed_at":"2026-08-03T09:40:31.365295Z","submitted_at":"2024-01-26T18:59:01Z","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15077","snapshot_observed_at":"2026-08-11T13:56:39.663532Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.663532Z"},"links":{"cited_paper":"/paper/2401.15077","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:1b378dfe93d737e03ae0e1a602fc12fdbe7b8b0843cce34352b4893a1cc926b1","observation_id":"018023f6-a511-40a7-abf7-378e706924c6","resolution":{"observed_at":"2026-08-11T13:56:39.663532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.421550Z","title":null,"venue":null,"work_id":"bc3bdbce-b3d5-4cd8-a0d8-af64bba1dd84","year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.670302Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:28ae8f1b059204e66f4d0c67955fc9f66c782e95520ff36c2553b93ef8852e6a","observation_id":"fd32a44b-b5b5-45b3-80be-770b28e89738","resolution":{"observed_at":"2026-08-11T13:56:40.426314Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.13581","last_updated":"2023-11-22T18:37:27Z","snapshot_observed_at":"2026-07-06T16:51:15.252642Z","submitted_at":"2023-11-22T18:37:27Z","title":"PaSS: Parallel Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.13581","snapshot_observed_at":"2026-08-11T13:56:39.674571Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.674571Z"},"links":{"cited_paper":"/paper/2311.13581","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:b3d0f3c829045a2828ee3ca5b64777bd3b90139b8f5ba7d6f545883bf3847f56","observation_id":"9c9e0826-8656-4623-89a3-6b8b1145ee52","resolution":{"observed_at":"2026-08-11T13:56:39.674571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-11T13:56:39.679327Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.679327Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:9600c4415f7de66c1050b85c6ad6eb062da13d1dfa6167e3a118738d2064131a","observation_id":"dc02a5c5-cb1a-44f5-b382-3ce5f730c8e1","resolution":{"observed_at":"2026-08-11T13:56:39.679327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.404189Z","title":null,"venue":null,"work_id":"0a99a7b9-bb9d-4069-a638-b3967b0614cb","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.688977Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:782af1f452321b5e510bc3348c01e88642081c6c3a8fbdac3c95c918e8dcab8a","observation_id":"85628b68-1705-4171-bacb-ad0d145d294f","resolution":{"observed_at":"2026-08-11T13:56:40.408816Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.383741Z","title":null,"venue":null,"work_id":"32d774b7-fdce-4040-99b5-edfa9a9b9b4e","year":2020},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.693837Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d007052bffdf7ccd260bfd2c4992969e68960744dd9a26896fe6d260ae4e895e","observation_id":"9885abe7-7969-4c1a-8386-d0338e15da0a","resolution":{"observed_at":"2026-08-11T13:56:40.388929Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.361580Z","title":null,"venue":null,"work_id":"585816b2-626c-40ce-9b55-4f40f7080e42","year":1992},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.699417Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:09340cdfabbb1d9e232b47d332d24627643394ad8024e2d2375bbe40fd69bbc6","observation_id":"d0236f5d-c9bd-4bdf-8d7c-04b16e9ac3d0","resolution":{"observed_at":"2026-08-11T13:56:40.369184Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.342621Z","title":null,"venue":null,"work_id":"e933b458-94d9-4b90-b16e-708d6476dc62","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.707155Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:14d91023a0d9ea8fb40cca09e6f83216b514c83782a7e4a0366e3b22a1e6871a","observation_id":"7ff372d0-6ce5-423d-9cf6-8092dfad27ca","resolution":{"observed_at":"2026-08-11T13:56:40.348487Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.324082Z","title":null,"venue":null,"work_id":"a3f390f7-1ef5-42b2-a20b-ed8fdbfad54a","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.713796Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d28ba9f3fb511f5ffbb3566b84d272c54b3a639ce5d584e8163b9c05a1766af9","observation_id":"dd547bb5-cfd0-4fcf-9315-2da5ce487123","resolution":{"observed_at":"2026-08-11T13:56:40.329362Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.04623","last_updated":"2023-08-08T23:29:55Z","snapshot_observed_at":"2026-08-09T22:39:17.904880Z","submitted_at":"2023-08-08T23:29:55Z","title":"Accelerating LLM Inference with Staged Speculative Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.04623","snapshot_observed_at":"2026-08-11T13:56:39.717832Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.717832Z"},"links":{"cited_paper":"/paper/2308.04623","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:881fb637d611add3baeb05cc1522b4c9d77f3517eb502d27a0a1323a4e5bc0b2","observation_id":"975475f5-45a8-4cae-b9be-ff813764f66e","resolution":{"observed_at":"2026-08-11T13:56:39.717832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.309008Z","title":null,"venue":null,"work_id":"955d4649-7c03-44f6-8260-8874a3d62258","year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.723725Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:b37478a72f4a192f7b6a300cce410fb8fc5a7f5bec97a66af98a7246b9371d54","observation_id":"209fa9cc-c15f-45fa-a7bc-36baa9148d3a","resolution":{"observed_at":"2026-08-11T13:56:40.313913Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.292599Z","title":null,"venue":null,"work_id":"04defcc9-002b-4e82-90ac-524e8b91dad2","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.729611Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:5fd78c15fc7b7665b110881cb9e791ce993fd4ae64f8933472431460586c5b1b","observation_id":"62d5b023-e51c-4dfe-a105-c319e0e4f94e","resolution":{"observed_at":"2026-08-11T13:56:40.298005Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1609.03499","last_updated":"2016-09-19T18:04:35Z","snapshot_observed_at":"2026-08-13T01:24:25.628327Z","submitted_at":"2016-09-12T17:29:40Z","title":"WaveNet: A Generative Model for Raw Audio","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.03499","snapshot_observed_at":"2026-08-11T13:56:39.734235Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.734235Z"},"links":{"cited_paper":"/paper/1609.03499","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d3c7abea46b3ba2844d38379323b8b694948c109ef875e151fcd77bf94a307d4","observation_id":"15dee101-6225-40c4-b916-5ceafb872b5e","resolution":{"observed_at":"2026-08-11T13:56:39.734235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.275240Z","title":null,"venue":null,"work_id":"de26a3bf-d67b-4633-a302-e2602091d7b6","year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.739625Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:ea274791fefdf883554b7672742ac2bf3b0df87402e3dc35843eaa83f9bc2f14","observation_id":"329f5adc-5855-4742-a60b-bdb427d69611","resolution":{"observed_at":"2026-08-11T13:56:40.280552Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03863","last_updated":"2024-05-23T06:08:37Z","snapshot_observed_at":"2026-08-02T01:09:26.116796Z","submitted_at":"2023-12-06T19:18:42Z","title":"Efficient Large Language Models: A Survey","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03863","snapshot_observed_at":"2026-08-11T13:56:39.744899Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.744899Z"},"links":{"cited_paper":"/paper/2312.03863","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:7560ce59d6bf50d0487cc8f3e70dcd0bd6782f305a0645e8193ef483c8f69658","observation_id":"162cc3db-c0d0-45ed-b514-07c4426e776d","resolution":{"observed_at":"2026-08-11T13:56:39.744899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.255618Z","title":null,"venue":null,"work_id":"d460a62b-eafe-4106-929c-3e996bc08021","year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.749541Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:34f2139e171ec2ec4748e540624fd698a77341786889bda129cb0061e20fe316","observation_id":"9f89cce7-c135-44d4-aa75-2e5694ae6d62","resolution":{"observed_at":"2026-08-11T13:56:40.260481Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.237391Z","title":null,"venue":null,"work_id":"3e4cfe43-cfeb-4ef4-83dd-13feb6e49f41","year":2019},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.756167Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:130251c727bf859ed2844189670cea4ace351888336e88cc1cf39c32a0833e9b","observation_id":"560487e9-aabd-4d9c-9d73-6e1b71cf50c8","resolution":{"observed_at":"2026-08-11T13:56:40.242349Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19124","last_updated":"2024-06-06T18:38:34Z","snapshot_observed_at":"2026-08-13T00:19:04.418458Z","submitted_at":"2024-04-29T21:59:07Z","title":"Accelerating Production LLMs with Combined Token/Embedding Speculators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19124","snapshot_observed_at":"2026-08-11T13:56:39.760297Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.760297Z"},"links":{"cited_paper":"/paper/2404.19124","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:cf6fdab6960399f8cf4433ef6f0975ecd9cd1fc9addb748d60adb147070308f4","observation_id":"399bc475-9fcd-49fc-b503-f18391603f7e","resolution":{"observed_at":"2026-08-11T13:56:39.760297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.213267Z","title":null,"venue":null,"work_id":"e2522118-c974-4af1-803c-f84fceba3955","year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.765552Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:67efc856b1e9262eb6e6c6b729033191a8fe1bb4055e7a5934529411b2b76f0b","observation_id":"9e19fcaf-326a-4171-89ea-ff63a7fc60b1","resolution":{"observed_at":"2026-08-11T13:56:40.217318Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-08-10T13:49:54.948163Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-11T13:56:39.772067Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.772067Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:f206e42eb7242c77bd94f10162c80c2b4e014647668a321467154974ea726980","observation_id":"6a36a5c1-318b-4044-b772-3322b34f97d5","resolution":{"observed_at":"2026-08-11T13:56:39.772067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.09269","last_updated":"2023-07-06T07:29:23Z","snapshot_observed_at":"2026-07-06T13:01:55.599089Z","submitted_at":"2022-04-20T07:25:22Z","title":"A Survey on Non-Autoregressive Generation for Neural Machine Translation and Beyond","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.09269","snapshot_observed_at":"2026-08-11T13:56:39.777354Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.777354Z"},"links":{"cited_paper":"/paper/2204.09269","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:737da03815758827c167aa01517b2fd538064a60c6e529d8f2ffa157591809a4","observation_id":"355893e3-66d0-4fdb-9c4b-b7356dafd000","resolution":{"observed_at":"2026-08-11T13:56:39.777354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:40.198755Z","title":null,"venue":null,"work_id":"14952ba5-f1b9-4ca4-b4d0-865a1e61ca7d","year":2021},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.782758Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:31787bb5a0365d55f4f273f37429aacdab3869754b56f3a8c5c79f3004195ed6","observation_id":"eecc80ce-032a-4d7f-a8c0-6b2e2fff821b","resolution":{"observed_at":"2026-08-11T13:56:40.202894Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08168","last_updated":"2024-05-20T02:37:20Z","snapshot_observed_at":"2026-08-12T23:34:26.836485Z","submitted_at":"2023-09-15T05:34:32Z","title":"Draft & Verify: Lossless Large Language Model Acceleration via Self-Speculative Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08168","snapshot_observed_at":"2026-08-11T13:56:39.788351Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.788351Z"},"links":{"cited_paper":"/paper/2309.08168","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:d041b86d24529643aa33af274d0636166f0950bb0edd2c8e22a78a97fb2d311c","observation_id":"368bf258-9b1c-4e01-a035-843ea2b7ba09","resolution":{"observed_at":"2026-08-11T13:56:39.788351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12728","last_updated":"2024-05-30T11:25:08Z","snapshot_observed_at":"2026-08-12T03:25:40.781473Z","submitted_at":"2023-12-20T02:55:15Z","title":"Lookahead: An Inference Acceleration Framework for Large Language Model with Lossless Generation Accuracy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12728","snapshot_observed_at":"2026-08-11T13:56:39.793794Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.793794Z"},"links":{"cited_paper":"/paper/2312.12728","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:3d61302ccd06ef8516501246ebb3ea2ab6af0cfdb370c38b9be7bb62e2928d32","observation_id":"871eb455-5dcf-4a0b-8ac6-99956511008e","resolution":{"observed_at":"2026-08-11T13:56:39.793794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:56:39.798834Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.798834Z"},"links":{"citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:6ed8a1f8d975efcfca4a812f6b1c772c23b08075ba35872f53a81faa8984db5a","observation_id":"1320dda8-e617-4b36-adfa-b0a7a0f8b86a","resolution":{"observed_at":"2026-08-11T13:56:39.798834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.07633","last_updated":"2024-07-30T13:14:55Z","snapshot_observed_at":"2026-08-06T06:16:16.015146Z","submitted_at":"2023-08-15T08:31:05Z","title":"A Survey on Model Compression for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.07633","snapshot_observed_at":"2026-08-11T13:56:39.803614Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.803614Z"},"links":{"cited_paper":"/paper/2308.07633","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:14b6b0b78e139b8651e8bd5417deaeea3e47fb9db5c94431b98d8904fa223d20","observation_id":"b884096a-6e81-40bd-ac00-1248f41f1541","resolution":{"observed_at":"2026-08-11T13:56:39.803614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.03382","last_updated":"2018-06-07T21:48:19Z","snapshot_observed_at":"2026-08-08T23:17:33.712950Z","submitted_at":"2018-03-09T04:39:35Z","title":"Fast Decoding in Sequence Models using Discrete Latent Variables","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.03382","snapshot_observed_at":"2026-08-11T13:56:39.808556Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T13:56:39.808556Z"},"links":{"cited_paper":"/paper/1803.03382","citing_paper":"/paper/2412.12639"},"observation_digest":"sha256:439a7c68c26574bed6a628211a1c652c6f22b2de92d948be002db0bdfb27100a","observation_id":"6908f2f6-4ebb-4d34-8076-e355979e08aa","resolution":{"observed_at":"2026-08-11T13:56:39.808556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.12639","last_updated":"2025-04-22T07:32:21Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-12T11:57:04.940808Z","submitted_at":"2024-12-17T08:02:08Z","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":2,"verified_fuzzy":2},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 1 inbound Pith citation observation for arXiv:2412.12639."}