{"as_of":"2026-08-06T22:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:673f595d55b9170880ba8aa73e24f222a64ca6515dfd9f3e515b2021fb39f8ae","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":10,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":10,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:06:48.536427Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T08:15:32.127896Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":"2502.01976","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-07-01T08:15:32.127896Z","title":"Citer: Collaborative inference for ef- ficient large language model decoding with token-level routing","venue":null,"work_id":"cf015044-91b5-45f2-ba27-ca0cb674052d","year":2025},"citing_paper":{"arxiv_id":"2502.18036","last_updated":"2026-04-22T02:19:48Z","snapshot_observed_at":"2026-08-02T22:41:05.099083Z","submitted_at":"2025-02-25T09:48:53Z","title":"Harnessing Multiple Large Language Models: A Survey on LLM Ensemble","version":6},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T02:22:28.649071Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2502.18036"},"observation_digest":"sha256:e2368844679bd52c22f46fff964abdc66714cf5690218c661ca38d866702594e","observation_id":"d945a911-9fa0-43ee-bd74-a244d66d22fa","resolution":{"observed_at":"2026-05-23T02:25:19.669400Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":"2502.01976","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-07-01T08:15:32.127896Z","title":"Citer: Collaborative inference for ef- ficient large language model decoding with token-level routing","venue":null,"work_id":"cf015044-91b5-45f2-ba27-ca0cb674052d","year":2025},"citing_paper":{"arxiv_id":"2506.14123","last_updated":"2026-05-07T07:43:36Z","snapshot_observed_at":"2026-07-06T21:43:22.202859Z","submitted_at":"2025-06-17T02:37:04Z","title":"Sampling from Your Language Model One Byte at a Time","version":3},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-19T09:52:20.923710Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2506.14123"},"observation_digest":"sha256:b7061996aac81b9aa7f86abfb004dcbbaa3a836292feba89310d61d3864b79c5","observation_id":"bb73bfbc-065e-45c1-8575-0053ad623da8","resolution":{"observed_at":"2026-05-19T09:53:02.078562Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-08-06T15:06:48.536427Z","title":"Citer: Collaborative inference for efficient large language model decoding with token-level routing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16731","last_updated":"2025-07-22T16:13:43Z","snapshot_observed_at":"2026-08-06T15:00:33.300567Z","submitted_at":"2025-07-22T16:13:43Z","title":"Collaborative Inference and Learning between Edge SLMs and Cloud LLMs: A Survey of Algorithms, Execution, and Open Challenges","version":1},"reference_index":183,"source":"pdf_text","source_observed_at":"2026-08-06T15:06:48.536427Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2507.16731"},"observation_digest":"sha256:331652431d69cc862a358006c51ded2b0b69e705b487cef4294bf337914b92aa","observation_id":"127465ed-f961-4ed3-8f79-c835b9b7616f","resolution":{"observed_at":"2026-08-06T15:06:48.536427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":"2502.01976","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-07-01T08:15:32.127896Z","title":"Citer: Collaborative inference for ef- ficient large language model decoding with token-level routing","venue":null,"work_id":"cf015044-91b5-45f2-ba27-ca0cb674052d","year":2025},"citing_paper":{"arxiv_id":"2604.18471","last_updated":"2026-04-26T18:45:11Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T16:22:59Z","title":"NI Sampling: Accelerating Discrete Diffusion Sampling by Token Order Optimization","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-10T05:56:52.205196Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2604.18471"},"observation_digest":"sha256:a29d40739439c7ef11b206bd53b2096f1aca4b12a7a77646a0b20b773375760d","observation_id":"2df8c9ca-e25c-4918-8ad8-efbfd166a930","resolution":{"observed_at":"2026-05-10T06:01:13.633017Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":"2502.01976","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-07-01T08:15:32.127896Z","title":"Citer: Collaborative inference for ef- ficient large language model decoding with token-level routing","venue":null,"work_id":"cf015044-91b5-45f2-ba27-ca0cb674052d","year":2025},"citing_paper":{"arxiv_id":"2605.00419","last_updated":"2026-05-25T08:32:35Z","snapshot_observed_at":"2026-08-03T02:25:15.458096Z","submitted_at":"2026-05-01T05:31:18Z","title":"Rethinking LLM Ensembling from the Perspective of Mixture Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-09T20:06:12.248439Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2605.00419"},"observation_digest":"sha256:8608bc31f8e3b871e4cf1f11fbeeb219c011c4187c79e58bcacd5a0808e56056","observation_id":"9482e410-3884-4ffd-b979-89f5c33f1c2a","resolution":{"observed_at":"2026-05-11T15:26:06.339051Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":"2502.01976","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-07-01T08:15:32.127896Z","title":"Citer: Collaborative inference for ef- ficient large language model decoding with token-level routing","venue":null,"work_id":"cf015044-91b5-45f2-ba27-ca0cb674052d","year":2025},"citing_paper":{"arxiv_id":"2605.00419","last_updated":"2026-05-25T08:32:35Z","snapshot_observed_at":"2026-08-03T02:25:15.458096Z","submitted_at":"2026-05-01T05:31:18Z","title":"Rethinking LLM Ensembling from the Perspective of Mixture Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-01T08:07:09.032028Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2605.00419"},"observation_digest":"sha256:6a51e476e10ae33496f359f200d07dc030aed34840e7dd77d589c593ef7577a1","observation_id":"1e596ca0-c9d6-4902-8889-de3a477d5455","resolution":{"observed_at":"2026-07-01T08:15:32.130199Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-08-02T15:18:54.607262Z","title":"CITER: Collaborative inference for efficient large language model decoding with token-level routing,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18244","last_updated":"2026-04-29T13:14:00Z","snapshot_observed_at":"2026-08-02T15:18:52.304452Z","submitted_at":"2026-04-29T13:14:00Z","title":"Accelerating Heterogeneous Agent Collaboration in Dynamic Edge Networks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T15:18:54.607262Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2607.18244"},"observation_digest":"sha256:1b8c267413c2dde31723817c37899b7ecf4f34a17a6162269417d955c612c73e","observation_id":"28555106-1416-4f58-b7f3-19a2b0208d0c","resolution":{"observed_at":"2026-08-02T15:18:54.607262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-08-01T10:14:15.660311Z","title":"Work in Progress 16 PyroDash: Cost-Efficient Token-Level Small-Large Model Collaborative Inference A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.20327","last_updated":"2026-07-22T16:14:26Z","snapshot_observed_at":"2026-08-06T13:03:05.099964Z","submitted_at":"2026-07-22T16:14:26Z","title":"PyroDash: Cost-Efficient Token-Level Small-Large Language Model Collaborative Inference","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T10:14:15.660311Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2607.20327"},"observation_digest":"sha256:144beeaf96727d9e660c1170767c207e37c1e357e30ce59e52df60cad9d34796","observation_id":"9791e570-20ef-4b91-9855-395ec8fb6379","resolution":{"observed_at":"2026-08-01T10:14:15.660311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-08-01T04:42:15.122849Z","title":"Citer: Collaborative inference for efficient large language model decoding with token-level routing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22465","last_updated":"2026-07-27T17:28:05Z","snapshot_observed_at":"2026-08-04T06:34:44.615698Z","submitted_at":"2026-07-24T16:29:06Z","title":"TRACE-ROUTER: Task-Consistent and Adaptive Online Routing for Agentic AI","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-01T04:42:15.122849Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2607.22465"},"observation_digest":"sha256:ed9aef044dac00f9f36dc9569c60eb4276d1baa3f32687a0a3b82a5719a321ae","observation_id":"07fd0cd4-105c-4dbf-bb4d-824d58d5dc52","resolution":{"observed_at":"2026-08-01T04:42:15.122849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01976","snapshot_observed_at":"2026-08-01T02:47:11.302112Z","title":"Citer: Collaborative inference for efficient large language model decoding with token-level routing.arXiv preprint arXiv:2502.01976, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27248","last_updated":"2026-07-28T06:52:19Z","snapshot_observed_at":"2026-08-06T03:42:29.828505Z","submitted_at":"2026-07-28T06:52:19Z","title":"Divergence Decoding: Training-Free Capability Fusion","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T02:47:11.302112Z"},"links":{"cited_paper":"/paper/2502.01976","citing_paper":"/paper/2607.27248"},"observation_digest":"sha256:aaa86a49b153bcc1569ca57d119d0c2b3d27568d39c37307c09ac186a3e8035a","observation_id":"c2049d7d-e5e1-44e9-9f02-3fe40069efe6","resolution":{"observed_at":"2026-08-01T02:47:11.302112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.01976/citation-record","integrity":"/paper/2502.01976/integrity","json":"/paper/2502.01976/citation-record.json","paper":"/paper/2502.01976"},"outbound":[],"paper":{"arxiv_id":"2502.01976","last_updated":"2025-09-10T02:45:51Z","latest_version":6,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T20:30:40.963009Z","submitted_at":"2025-02-04T03:36:44Z","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 10 inbound Pith citation observations for arXiv:2502.01976."}