{"as_of":"2026-08-08T11:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0739bcc255fa80c6087b8b7c6481c0470d58a7bc937b49197eb7e8160710095c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:56:43.616119Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":106,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2211.05100","last_updated":"2023-06-27T09:57:58Z","snapshot_observed_at":"2026-08-04T18:56:03.233715Z","submitted_at":"2022-11-09T18:48:09Z","title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","version":4},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-05-12T00:51:10.919818Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2211.05100"},"observation_digest":"sha256:0fc2b3bd8a79314afedb334ede15c6b7c06c7a1b10581da527a299ca7c1a1ba1","observation_id":"7a1c99ac-44f9-4d62-887c-8c9e45f07472","resolution":{"observed_at":"2026-05-12T00:51:11.075762Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2211.05100","last_updated":"2023-06-27T09:57:58Z","snapshot_observed_at":"2026-08-04T18:56:03.233715Z","submitted_at":"2022-11-09T18:48:09Z","title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","version":4},"reference_index":278,"source":"arxiv_source","source_observed_at":"2026-05-12T00:51:10.919818Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2211.05100"},"observation_digest":"sha256:26c226cd528e9f72bd836e3b573f88faa7268b3c1851b950a555adc772977f61","observation_id":"f105b10b-2fd9-4768-bee4-7a87a9141599","resolution":{"observed_at":"2026-05-12T00:51:11.646731Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2303.17564","last_updated":"2023-12-21T06:21:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-30T17:30:36Z","title":"BloombergGPT: A Large Language Model for Finance","version":3},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-13T23:19:46.231145Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2303.17564"},"observation_digest":"sha256:2ffa8376a31929ec129e7f2b8e02b60414ee7695fc11c5bda4dd7f73d680e247","observation_id":"c204e047-8388-4312-9431-0d41cc1ddd0b","resolution":{"observed_at":"2026-05-13T23:19:46.729797Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2307.06435","last_updated":"2024-10-17T01:10:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-12T20:01:52Z","title":"A Comprehensive Overview of Large Language Models","version":10},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-19T20:28:38.900026Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2307.06435"},"observation_digest":"sha256:f72fd5e7bb59e2d58d392fb49a10e38b0db4e579f9d27f4a995e194c8f80b363","observation_id":"d570476c-b086-43ce-89e5-2fdbafa6d6fa","resolution":{"observed_at":"2026-05-19T20:32:45.607604Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2403.19887","last_updated":"2024-07-03T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-28T23:55:06Z","title":"Jamba: A Hybrid Transformer-Mamba Language Model","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T14:11:27.156350Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2403.19887"},"observation_digest":"sha256:159f47062e3bf5bfe982db670decc56b9d56f12c9e336f74d1081e2a8fb011fc","observation_id":"4bd88b63-8ba7-4bb3-a279-175228a7692a","resolution":{"observed_at":"2026-05-13T14:11:27.200041Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-07T14:56:43.616119Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.16868","last_updated":"2025-05-22T16:24:37Z","snapshot_observed_at":"2026-08-07T14:51:41.178935Z","submitted_at":"2025-05-22T16:24:37Z","title":"Comparative analysis of subword tokenization approaches for Indian languages","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:56:43.616119Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2505.16868"},"observation_digest":"sha256:bcf0540d9cb7c5c26b8cf74667d61b40e3c1c21cf2b444ed5cf569b33bef5811","observation_id":"4f013cee-8e99-4990-b05f-77b42f49ad0b","resolution":{"observed_at":"2026-08-07T14:56:43.616119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-07T11:18:11.036474Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gallé, Arun Raja, Chenglei Si, Wilson Y","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.03101","last_updated":"2025-06-03T17:35:56Z","snapshot_observed_at":"2026-08-08T07:38:04.360811Z","submitted_at":"2025-06-03T17:35:56Z","title":"Beyond Text Compression: Evaluating Tokenizers Across Scales","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-07T11:18:11.036474Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2506.03101"},"observation_digest":"sha256:993a23060ded33275b82b28cbb01a9d5675c9235f482beb7ad6da65327835a1b","observation_id":"1f1860a8-eae6-4cc5-b178-70030171147b","resolution":{"observed_at":"2026-08-07T11:18:11.036474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-07T10:54:38.211148Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.04032","last_updated":"2025-06-04T14:56:08Z","snapshot_observed_at":"2026-08-07T10:46:35.682379Z","submitted_at":"2025-06-04T14:56:08Z","title":"AI Agents for Conversational Patient Triage: Preliminary Simulation-Based Evaluation with Real-World EHR Data","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T10:54:38.211148Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2506.04032"},"observation_digest":"sha256:a6a4e57082edeb89da35b937c06859d18fc9956dc7d9b0863a49a07207bf3003","observation_id":"bde9fb3c-eb54-42c1-b85b-04165b01106a","resolution":{"observed_at":"2026-08-07T10:54:38.211148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-07T05:40:47.776405Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.07541","last_updated":"2025-06-09T08:28:16Z","snapshot_observed_at":"2026-08-07T05:28:59.548607Z","submitted_at":"2025-06-09T08:28:16Z","title":"Bit-level BPE: Below the byte boundary","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T05:40:47.776405Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2506.07541"},"observation_digest":"sha256:ffb9dca01c9d4681ed302e5c8260e68a46b4d2afa428d82e9027357706a9709c","observation_id":"f176c6cb-890a-4371-b7b6-e53edebd78bf","resolution":{"observed_at":"2026-08-07T05:40:47.776405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-06T23:19:15.192201Z","title":"arXiv preprint arXiv:2112.10508","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.18674","last_updated":"2025-06-23T14:18:46Z","snapshot_observed_at":"2026-08-06T23:13:17.444822Z","submitted_at":"2025-06-23T14:18:46Z","title":"Is There a Case for Conversation Optimized Tokenizers in Large Language Models?","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T23:19:15.192201Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2506.18674"},"observation_digest":"sha256:05df39d7500461aef6c6ca024d9200cf7cdc093b2e8c5e35213a08e4c332e42c","observation_id":"c4f9ba5f-e3f0-4321-a224-fd5e3daa375c","resolution":{"observed_at":"2026-08-06T23:19:15.192201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-06T22:46:40.446794Z","title":"Between words and characters: A brief history of open-vocabulary modeling and tokenization in nlp","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.20920","last_updated":"2025-06-26T01:01:47Z","snapshot_observed_at":"2026-08-07T23:54:52.849436Z","submitted_at":"2025-06-26T01:01:47Z","title":"FineWeb2: One Pipeline to Scale Them All -- Adapting Pre-Training Data Processing to Every Language","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T22:46:40.446794Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2506.20920"},"observation_digest":"sha256:ac80f34d71da18dd723127391d8f8db99e17145684ef7490a7291232a02aaaae","observation_id":"30ff154b-ccfc-425f-815b-7b6da12380b0","resolution":{"observed_at":"2026-08-06T22:46:40.446794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T22:43:23.055609Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gallé, Arun Raja, Chenglei Si, Wilson Y","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.06621","last_updated":"2025-08-08T18:10:03Z","snapshot_observed_at":"2026-08-06T23:15:18.955269Z","submitted_at":"2025-08-08T18:10:03Z","title":"Train It and Forget It: Merge Lists are Unnecessary for BPE Inference in Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T22:43:23.055609Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2508.06621"},"observation_digest":"sha256:20334c62e71cbab4137d1fa02d5385f47c31f249ae3d2666b06e5b8b08cf11f3","observation_id":"a3c0260e-6aa5-4a87-8872-85a002c12540","resolution":{"observed_at":"2026-08-05T22:43:23.055609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T15:26:02.073354Z","title":"J.; Alyafeai, Z.; Salesky, E.; Raffel, C.; Dey, M.; Gall \\'e , M.; Raja, A.; Si, C.; Lee, W","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.19924","last_updated":"2025-08-27T14:32:15Z","snapshot_observed_at":"2026-08-06T03:02:37.512532Z","submitted_at":"2025-08-27T14:32:15Z","title":"FlowletFormer: Network Behavioral Semantic Aware Pre-training Model for Traffic Classification","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T15:26:02.073354Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2508.19924"},"observation_digest":"sha256:bb39e510f6721e0539afffdd7e687e13aebc00cb852ab9b74b2f64caf295cfe9","observation_id":"23e77333-f6c1-4321-9e02-8bce1b0b9835","resolution":{"observed_at":"2026-08-05T15:26:02.073354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T13:26:15.129449Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00675","last_updated":"2025-08-31T03:06:37Z","snapshot_observed_at":"2026-08-08T00:24:33.198372Z","submitted_at":"2025-08-31T03:06:37Z","title":"Speaker-Conditioned Phrase Break Prediction for Text-to-Speech with Phoneme-Level Pre-trained Language Model","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T13:26:15.129449Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2509.00675"},"observation_digest":"sha256:10be85feeb033f9c00f98819092dd8b64880f0c35e1b633e74d29180fdd2c3f3","observation_id":"c7abf7d9-1473-425e-b7cf-f7e24b68acfd","resolution":{"observed_at":"2026-08-05T13:26:15.129449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T11:01:20.305515Z","title":"Mielke, Z","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.03407","last_updated":"2025-09-03T15:32:50Z","snapshot_observed_at":"2026-08-08T03:54:24.867080Z","submitted_at":"2025-09-03T15:32:50Z","title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T11:01:20.305515Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2509.03407"},"observation_digest":"sha256:c5299d66758b105e0f7346a26a4928285f2a0ce23d2a8b2b6cb821fcc3a71340","observation_id":"abb38b1c-7949-4d7b-94ac-cfc58ebfb653","resolution":{"observed_at":"2026-08-05T11:01:20.305515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2604.14171","last_updated":"2026-03-25T07:02:51Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-03-25T07:02:51Z","title":"Benchmarking Linguistic Adaptation in Comparable-Sized LLMs: A Study of Llama-3.1-8B, Mistral-7B-v0.1, and Qwen3-8B on Romanized Nepali","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T00:48:32.432546Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2604.14171"},"observation_digest":"sha256:d188d7f1191b8fffd378863d9966df18795c7957e617e9d324153a05a8fb27f2","observation_id":"aeb96fa4-aa26-40e8-9ff8-adabf20151cb","resolution":{"observed_at":"2026-05-15T00:49:36.205280Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2604.18423","last_updated":"2026-04-20T15:41:05Z","snapshot_observed_at":"2026-07-06T23:05:17.639285Z","submitted_at":"2026-04-20T15:41:05Z","title":"BhashaSutra: A Task-Centric Unified Survey of Indian NLP Datasets, Corpora, and Resources","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T04:22:34.046014Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2604.18423"},"observation_digest":"sha256:2061a95d810a1e6d40d15ce93db6f7f5777bc63f1b1fe7f8a23444ebc7126eb2","observation_id":"a2515f8e-e534-46a0-8c5e-42044d551339","resolution":{"observed_at":"2026-05-11T12:01:02.017768Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2605.17152","last_updated":"2026-05-16T20:56:15Z","snapshot_observed_at":"2026-08-02T18:07:50.795904Z","submitted_at":"2026-05-16T20:56:15Z","title":"Multilingual and Multimodal LLMs in the Wild: Building for Low-Resource Languages","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-20T14:33:36.100966Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2605.17152"},"observation_digest":"sha256:0c476add8ebb81800694e192aad55c742aae4437c37c164ed1bac5a415258cbd","observation_id":"b47d1164-f5aa-4c99-91d0-46eeade1d351","resolution":{"observed_at":"2026-05-20T14:38:21.835952Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2605.22403","last_updated":"2026-05-21T12:31:47Z","snapshot_observed_at":"2026-08-01T16:03:13.526774Z","submitted_at":"2026-05-21T12:31:47Z","title":"Translating Signals to Languages for sEMG-Based Activity Recognition","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-22T07:41:26.887209Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2605.22403"},"observation_digest":"sha256:f59329b2e0833404a76a955050068b6d1e8fa11ecfc0c18ff500d3bbc0f40438","observation_id":"93d585ba-5dcb-4b0c-8628-a416be556035","resolution":{"observed_at":"2026-05-22T07:44:42.857609Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2605.29459","last_updated":"2026-05-28T06:53:18Z","snapshot_observed_at":"2026-08-06T01:09:11.993125Z","submitted_at":"2026-05-28T06:53:18Z","title":"Kronecker Embeddings: Byte-Level Structured Token Representations for Parameter-Efficient Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T07:50:25.019889Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2605.29459"},"observation_digest":"sha256:c450a2d382881f908e87f6e2bd1f4ee4f04faa4480c56f33c9b31bbe825909b3","observation_id":"533da98c-0bd8-4f52-a6a3-7915e716954c","resolution":{"observed_at":"2026-06-29T07:53:13.399554Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2606.08713","last_updated":"2026-06-07T16:09:04Z","snapshot_observed_at":"2026-08-02T22:19:02.442419Z","submitted_at":"2026-06-07T16:09:04Z","title":"The price of incrementality in k-center clustering","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-06-27T17:41:41.502995Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2606.08713"},"observation_digest":"sha256:c808f65f5352ffe917c387b2910b23f073dcebf6d9de15a6dd708a31830b96db","observation_id":"85e7a8d4-9ca4-40a2-9acf-a8dfb0cd6e7d","resolution":{"observed_at":"2026-07-02T23:57:28.310361Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2606.20993","last_updated":"2026-06-18T23:50:54Z","snapshot_observed_at":"2026-08-05T09:19:18.829577Z","submitted_at":"2026-06-18T23:50:54Z","title":"Phonemes to the Rescue: Multilingual Tokenization Based on International Phonetic Alphabet","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-06-26T16:53:25.556222Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2606.20993"},"observation_digest":"sha256:b3be0f0a1f93fa0fcec90821f2a15b464ef26b03aaa4d406c169a53ab1b26ef0","observation_id":"cab02254-afc5-4f27-95c4-0ee3846bfd24","resolution":{"observed_at":"2026-07-04T04:39:34.660865Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","version":1},"cited_work":{"arxiv_id":"2112.10508","doi":"10.48550/arxiv.2112.10508","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.10508","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mielke, Zaid Alyafeai, Elizabeth Salesky, Colin Raffel, Manan Dey, Matthias Gall ´e, Arun Raja, Chen- glei Si, Wilson Y","venue":"arXiv (Cornell University)","work_id":"89e0e401-f1b4-4642-afbc-b411df824e3d","year":2021},"citing_paper":{"arxiv_id":"2606.27019","last_updated":"2026-06-25T13:31:02Z","snapshot_observed_at":"2026-07-07T00:01:15.803410Z","submitted_at":"2026-06-25T13:31:02Z","title":"MinGram: A Minimalist Unigram Tokenizer with High Compression and Competitive Morphological Alignment","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-06-26T04:59:05.057552Z"},"links":{"cited_paper":"/paper/2112.10508","citing_paper":"/paper/2606.27019"},"observation_digest":"sha256:3957a6742f9db9d9611a29e0ca8d9dd2e801aa61bececa85db559f15f1dc0c86","observation_id":"60b98f86-83b7-4010-af89-a83726b84097","resolution":{"observed_at":"2026-07-04T13:39:51.297220Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2112.10508/citation-record","integrity":"/paper/2112.10508/integrity","json":"/paper/2112.10508/citation-record.json","paper":"/paper/2112.10508"},"outbound":[],"paper":{"arxiv_id":"2112.10508","last_updated":"2021-12-20T13:04:18Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T12:20:36.165718Z","submitted_at":"2021-12-20T13:04:18Z","title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2112.10508."}