{"as_of":"2026-08-17T14:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9127aaa92c9019f47fe5c9780739ddfd69616977b1dbb233f9fbddf39411fe64","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T21:30:03.795718Z","state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:25:13.091782Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T00:38:54.307688Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09738","snapshot_observed_at":"2026-08-07T04:25:13.091782Z","title":"Noam Shazeer","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.10766","last_updated":"2025-06-12T14:47:13Z","snapshot_observed_at":"2026-08-15T10:34:26.341703Z","submitted_at":"2025-06-12T14:47:13Z","title":"One Tokenizer To Rule Them All: Emergent Language Plasticity via Multilingual Tokenizers","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T04:25:13.091782Z"},"links":{"cited_paper":"/paper/2505.09738","citing_paper":"/paper/2506.10766"},"observation_digest":"sha256:b71eb8ce12e4790b0ce9f88dc36c0b39e23fac355c25582cee5be188a9c62a9b","observation_id":"744632db-d0d0-4464-9d5d-2d4e7ddb6978","resolution":{"observed_at":"2026-08-07T04:25:13.091782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09738","snapshot_observed_at":"2026-08-02T22:58:10.690941Z","title":"Achieving tokenizer flexibility in language models through heuristic adaptation and supertoken learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.15382","last_updated":"2026-05-28T08:46:27Z","snapshot_observed_at":"2026-08-16T04:02:24.978329Z","submitted_at":"2026-02-17T06:31:53Z","title":"The Vision Wormhole: Latent-Space Communication in Heterogeneous Multi-Agent Systems","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-02T22:58:10.690941Z"},"links":{"cited_paper":"/paper/2505.09738","citing_paper":"/paper/2602.15382"},"observation_digest":"sha256:3d85ccb9818b1b13b29e68a572c44bf13bba3750a787fa9ed7d14ef4833c8682","observation_id":"e40915e6-15aa-48d3-b672-1eb04ce95436","resolution":{"observed_at":"2026-08-02T22:58:10.690941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09738","snapshot_observed_at":"2026-08-01T23:50:17.166923Z","title":"Achieving tokenizer flexibility in language models through heuristic adaptation and supertoken learning.arXiv preprint arXiv:2505.09738,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15232","last_updated":"2026-07-16T17:32:38Z","snapshot_observed_at":"2026-08-15T16:23:09.297692Z","submitted_at":"2026-07-16T17:32:38Z","title":"In-Place Tokenizer Expansion for Pre-trained LLMs","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-01T23:50:17.166923Z"},"links":{"cited_paper":"/paper/2505.09738","citing_paper":"/paper/2607.15232"},"observation_digest":"sha256:596f9453c7da0fc4e50a775eb81063b5c3c9263fa2381d42475e54fd6afbb7ae","observation_id":"566d3c34-cc63-4f5f-b2ea-a9eae8368a52","resolution":{"observed_at":"2026-08-01T23:50:17.166923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"cited_work":{"arxiv_id":"2505.09738","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.09738","snapshot_observed_at":"2026-08-05T00:38:54.307688Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","venue":"cs.CL","work_id":"972be7ac-143e-45cb-a390-652c1f3fe4a6","year":2025},"citing_paper":{"arxiv_id":"2608.00582","last_updated":"2026-08-01T10:38:23Z","snapshot_observed_at":"2026-08-12T01:52:34.240132Z","submitted_at":"2026-08-01T10:38:23Z","title":"Writing-System-Level Tokenizer Adaptation for Byte-Level BPE","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-05T00:38:53.782733Z"},"links":{"cited_paper":"/paper/2505.09738","citing_paper":"/paper/2608.00582"},"observation_digest":"sha256:09866b05242ef69cac2ba521f268a37f0ca3a06addc2162f7d397ff37fb8cff2","observation_id":"a3175c71-e39c-4f06-bddb-def2e6391fd0","resolution":{"observed_at":"2026-08-05T00:38:54.418329Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.09738/citation-record","integrity":"/paper/2505.09738/integrity","json":"/paper/2505.09738/citation-record.json","paper":"/paper/2505.09738"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.04335","last_updated":"2024-10-06T03:01:07Z","snapshot_observed_at":"2026-08-16T13:12:14.759853Z","submitted_at":"2024-10-06T03:01:07Z","title":"ReTok: Replacing Tokenizer to Enhance Representation Efficiency in Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04335","snapshot_observed_at":"2026-08-15T21:30:03.740361Z","title":"codeparrot","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.740361Z"},"links":{"cited_paper":"/paper/2410.04335","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:e6e23433eaead539e3ec79250d00dbf18afcc43a483b310ecd3c03a3486718a0","observation_id":"4e2aa0f6-8172-4036-88b6-6c6167d80146","resolution":{"observed_at":"2026-08-15T21:30:03.740361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10712","last_updated":"2024-09-26T11:15:14Z","snapshot_observed_at":"2026-08-16T14:17:58.381351Z","submitted_at":"2024-02-16T14:15:15Z","title":"An Empirical Study on Cross-lingual Vocabulary Adaptation for Efficient Language Model Inference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10712","snapshot_observed_at":"2026-08-15T21:30:03.748757Z","title":"Jesse Dodge, Maarten Sap, Ana Marasovi, William Agnew, Gabriel Ilharco, Dirk Groeneveld, Margaret Mitchell, and Matt Gardner","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.748757Z"},"links":{"cited_paper":"/paper/2402.10712","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:f27552f2bf9ea506d294d930150cf033dfc38ea55a3db0368c575bf24061f0a2","observation_id":"22befeec-313f-4062-ac01-362ad341eb28","resolution":{"observed_at":"2026-08-15T21:30:03.748757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08758","last_updated":"2021-09-30T17:20:01Z","snapshot_observed_at":"2026-08-16T18:30:48.108993Z","submitted_at":"2021-04-18T07:42:52Z","title":"Documenting Large Webtext Corpora: A Case Study on the Colossal Clean Crawled Corpus","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08758","snapshot_observed_at":"2026-08-15T21:30:03.753278Z","title":"Jay Gala, Thanmay Jayakumar, Jaavid Aktar Husain, Aswanth Kumar M, Mohammed Safi Ur Rahman Khan, Diptesh Kanojia, Ratish Puduppully, Mitesh M","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.753278Z"},"links":{"cited_paper":"/paper/2104.08758","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:15bd33e55e9489e4dd48c2d5eec17830b6165a0e2464c18269dc02d1f12372f5","observation_id":"73b21c49-599b-4ed1-a22f-805c7da9fc19","resolution":{"observed_at":"2026-08-15T21:30:03.753278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15006","last_updated":"2024-02-26T12:17:25Z","snapshot_observed_at":"2026-08-16T14:24:10.832214Z","submitted_at":"2024-01-26T17:07:08Z","title":"Airavata: Introducing Hindi Instruction-tuned LLM","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15006","snapshot_observed_at":"2026-08-15T21:30:03.758013Z","title":"Andrew R","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.758013Z"},"links":{"cited_paper":"/paper/2401.15006","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:3059311c5aaa66206e878ca28aaee9c8b86727f6969cd802e39d563fd1850083","observation_id":"dd926726-e4b4-429c-bd75-a986b2989689","resolution":{"observed_at":"2026-08-15T21:30:03.758013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04303","last_updated":"2024-08-08T08:37:28Z","snapshot_observed_at":"2026-08-16T13:27:45.567095Z","submitted_at":"2024-08-08T08:37:28Z","title":"Trans-Tokenization and Cross-lingual Vocabulary Transfers: Language Adaptation of LLMs for Low-Resource NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04303","snapshot_observed_at":"2026-08-15T21:30:03.762381Z","title":"Suchin Gururangan, Ana Marasović, Swabha Swayamdipta, Kyle Lo, Iz Beltagy, Doug Downey, and Noah A","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.762381Z"},"links":{"cited_paper":"/paper/2408.04303","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:ac72b4b1c73287bd513d7f76aabd2219d047b0d583bd3fbc23ab5f9b1605d94a","observation_id":"e6088e76-f291-4698-8aa2-21c1e9409625","resolution":{"observed_at":"2026-08-15T21:30:03.762381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06598","last_updated":"2022-05-04T08:53:32Z","snapshot_observed_at":"2026-08-16T17:35:08.636663Z","submitted_at":"2021-12-13T12:26:02Z","title":"WECHSEL: Effective initialization of subword embeddings for cross-lingual transfer of monolingual language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06598","snapshot_observed_at":"2026-08-15T21:30:03.774027Z","title":"Oscar Minixhofer, Gábor Berend, and Jie Yang","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.774027Z"},"links":{"cited_paper":"/paper/2112.06598","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:716cd1225148a76bcc2105b78df6058888b73ec25e1cb9f92336f72810c980a9","observation_id":"6239edb4-55f0-4240-ae0b-84260c0ed6d1","resolution":{"observed_at":"2026-08-15T21:30:03.774027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08966","last_updated":"2024-12-13T04:59:10Z","snapshot_observed_at":"2026-08-16T18:45:59.806789Z","submitted_at":"2024-02-14T06:20:48Z","title":"Pretraining Vision-Language Model for Difference Visual Question Answering in Longitudinal Chest X-rays","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08966","snapshot_observed_at":"2026-08-15T21:30:03.777814Z","title":"org/abs/2402.08966","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.777814Z"},"links":{"cited_paper":"/paper/2402.08966","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:9dceefebdd3a3919aa02fbdb2cc7f9ef05e4b302401b9b765a2f396d0e81d1aa","observation_id":"8ac63071-f400-43c9-9dec-3435350947f5","resolution":{"observed_at":"2026-08-15T21:30:03.777814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.03608","last_updated":"2024-01-07T23:35:48Z","snapshot_observed_at":"2026-08-16T14:29:15.983101Z","submitted_at":"2024-01-07T23:35:48Z","title":"The ultraspherical rectangular collocation method and its convergence","version":1},"cited_work":{"arxiv_id":"2401.03608","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.03608","snapshot_observed_at":"2026-08-15T21:30:03.871271Z","title":"The ultraspherical rectangular collocation method and its convergence","venue":"math.NA","work_id":"f96ba7bd-402c-418e-9523-90b63278a664","year":2024},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.786577Z"},"links":{"cited_paper":"/paper/2401.03608","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:0f6d720eef8604dc11ef834c95b4e061cedc7ea8be37ed6568545e29979428c4","observation_id":"be618465-4cbb-41db-b90e-2ecc74ea96af","resolution":{"observed_at":"2026-08-15T21:30:03.877668Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-08-17T01:19:18.409791Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-08-15T21:30:03.791522Z","title":"Yifan Zhang, Yifan Luo, Yang Yuan, and Andrew Chi-Chih Yao","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.791522Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:58420cc4b2d89f587d2d826f82e1751cbe8755c7364b9d8ebf4ffd7c80240e35","observation_id":"b0a10a52-e42c-40da-95eb-fcbe414bd2c3","resolution":{"observed_at":"2026-08-15T21:30:03.791522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07625","last_updated":"2025-07-22T09:17:48Z","snapshot_observed_at":"2026-08-16T23:13:41.659759Z","submitted_at":"2024-02-12T13:09:21Z","title":"Autonomous Data Selection with Zero-shot Generative Classifiers for Mathematical Texts","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07625","snapshot_observed_at":"2026-08-15T21:30:03.795718Z","title":"supertokens","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.795718Z"},"links":{"cited_paper":"/paper/2402.07625","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:e572ed010fe8f4f278bc5596134621249f44d0d508a995d936e864f02361b132","observation_id":"9b6c678e-1e54-4153-b26a-4cde16206477","resolution":{"observed_at":"2026-08-15T21:30:03.795718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:30:03.782214Z","title":"doi: 10.18653/v1/ P16-1162","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.782214Z"},"links":{"citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:089e2a7167f25b9e089b44b8540a36cea294e5b1cebf07ae82231b2b9ecdd874","observation_id":"315fa27d-32fe-4ab9-88b4-804ca606584a","resolution":{"observed_at":"2026-08-15T21:30:03.782214Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:30:04.012173Z","title":null,"venue":null,"work_id":"a8cb4705-54f5-4e25-9a97-886accf646cc","year":2004},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.766499Z"},"links":{"citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:d26a6feb2bf6886e176af96db2e917d2df915aa181d57b2ebd7d36728c0e3a7d","observation_id":"67aba5da-2eeb-4886-8c77-8bd1c2247c84","resolution":{"observed_at":"2026-08-15T21:30:04.015946Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.05845","last_updated":"2023-11-10T03:02:39Z","snapshot_observed_at":"2026-08-17T13:00:18.340998Z","submitted_at":"2023-11-10T03:02:39Z","title":"Tamil-Llama: A New Tamil Language Model Based on Llama 2","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.05845","snapshot_observed_at":"2026-08-15T21:30:03.736366Z","title":"ZhenChen,JianingWang,QiushiSun,XiuboGeng,NuoXu,WenjiMao,andDaxinJiang","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.736366Z"},"links":{"cited_paper":"/paper/2311.05845","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:1ea837e7ffe7cf7da9d765a8e373c5c116035151c489e8eb23add7a70339c034","observation_id":"644ef3b4-a3a6-4d1c-9612-bfde45b66f07","resolution":{"observed_at":"2026-08-15T21:30:03.736366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10244","last_updated":"2024-02-15T04:25:50Z","snapshot_observed_at":"2026-08-16T14:18:26.282468Z","submitted_at":"2024-02-15T04:25:50Z","title":"Entanglement generation in capacitively coupled Transmon-cavity system","version":1},"cited_work":{"arxiv_id":"2402.10244","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.10244","snapshot_observed_at":"2026-08-15T21:30:03.999238Z","title":"Entanglement generation in capacitively coupled Transmon-cavity system","venue":"quant-ph","work_id":"7bdba2d3-642e-4e8f-b3e4-e0601a222570","year":2024},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.731822Z"},"links":{"cited_paper":"/paper/2402.10244","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:3e185997d4ca09e3f5d2413b2a3a404499bc308b6124494f2b0a307b549b5b18","observation_id":"07b60db6-c8cc-41de-b76e-b130947d0468","resolution":{"observed_at":"2026-08-15T21:30:04.003902Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13423","last_updated":"2025-08-26T18:43:00Z","snapshot_observed_at":"2026-08-16T12:49:15.652378Z","submitted_at":"2025-03-17T17:53:23Z","title":"SuperBPE: Space Travel for Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13423","snapshot_observed_at":"2026-08-15T21:30:03.770462Z","title":"Meta-Llama","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-15T21:30:03.770462Z"},"links":{"cited_paper":"/paper/2503.13423","citing_paper":"/paper/2505.09738"},"observation_digest":"sha256:560a6c2f4fbc2dd466b4c00d7695fbe79641245c8f0c1dcae35fba337b78f01b","observation_id":"37261bbf-4ae6-4b7e-a584-ed7e70e96505","resolution":{"observed_at":"2026-08-15T21:30:03.770462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.09738","last_updated":"2025-05-14T19:00:27Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-15T21:23:24.645855Z","submitted_at":"2025-05-14T19:00:27Z","title":"Achieving Tokenizer Flexibility in Language Models through Heuristic Adaptation and Supertoken Learning"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":12,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 4 inbound Pith citation observations for arXiv:2505.09738."}