{"as_of":"2026-08-06T22:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bb758e9ea5edfa2ccbaefa2205e1a62d7345768618e0a57daf61d49454e0b572","coverage":[{"denominator":104,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T04:05:28.713898Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.09630/citation-record","integrity":"/paper/2605.09630/integrity","json":"/paper/2605.09630/citation-record.json","paper":"/paper/2605.09630"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.08818","last_updated":"2024-11-17T00:41:01Z","snapshot_observed_at":"2026-08-03T11:13:52.159847Z","submitted_at":"2024-07-11T18:59:21Z","title":"MAGNET: Improving the Multilingual Fairness of Language Models with Adaptive Gradient-Based Tokenization","version":2},"cited_work":{"arxiv_id":"2407.08818","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.08818","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"emnlp-main.614/","venue":null,"work_id":"c19a5144-aef7-4515-a0d5-05fde00aa9e8","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2407.08818","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:0812ed21500f0db91055b2b125efaf8a3e597935fed42a37676bd24c5f508ab5","observation_id":"a5b37670-bc1b-42e1-9b35-d2fcf54fbf07","resolution":{"observed_at":"2026-05-12T06:36:29.152842Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Character-level language modeling with deeper self-attention","venue":null,"work_id":"8432ab65-c214-44e1-a016-db2b1d31e420","year":2019},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:29c23e1832355a15d6041376440f9056ca9bf7cd1b0feeada6e4250600f0e3a9","observation_id":"df4ae625-4042-4327-b252-9c5c2f7e1cd0","resolution":{"observed_at":"2026-05-12T16:46:42.477886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":"2108.07732","doi":"10.1007/s11390-025-5518-5","metadata_source":"pith","pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Program Synthesis with Large Language Models","venue":"cs.PL","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:a9d967d630f80921a5a6ab8d306d47a177eb39d8da273288b5f636ecc77b58ac","observation_id":"d39d824f-347c-4ed5-8e2d-21940bc8b05e","resolution":{"observed_at":"2026-05-12T06:36:29.307355Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T17:02:41.000978Z","title":"Relaxed recursive transformers: Effective parameter sharing with layer-wise lo RA","venue":null,"work_id":"1c7223ff-e653-421c-90b4-bcdf539d85f1","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:6bf9d3ff6952393f13c666c8a547d637aeaa4ccefeac699cee621f000b34e39b","observation_id":"bed9bb2b-1293-40a8-904e-24184a3745c3","resolution":{"observed_at":"2026-05-12T16:46:42.446606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T17:02:40.989006Z","title":"Mixture-of-recursions: Learning dynamic recursive depths for adaptive token-level computation","venue":null,"work_id":"7b908f62-00e5-49c6-ac4f-bef5e029cd84","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:64ea28541b7ea543fab5c31bc37e23296c001de018291cd38433853da1377187","observation_id":"4ccde85b-60f6-435b-b7af-0840607b71f9","resolution":{"observed_at":"2026-05-12T16:46:42.489276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pondernet: Learning to ponder","venue":null,"work_id":"343d06f5-8478-446d-8cea-1339f5d8b93f","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:eb0119a79c52e8a0a4d635bb7886309efbdab89148a0d74eabac003579804b57","observation_id":"eafd033d-358c-4001-8d82-9f8e21d16ff0","resolution":{"observed_at":"2026-05-12T16:46:42.318848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08821","last_updated":"2024-12-15T21:20:12Z","snapshot_observed_at":"2026-08-06T10:11:02.511531Z","submitted_at":"2024-12-11T23:36:20Z","title":"Large Concept Models: Language Modeling in a Sentence Representation Space","version":2},"cited_work":{"arxiv_id":"2412.08821","doi":"10.48550/arxiv.2412.08821","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.08821","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large concept models: Language modeling in a sentence representation space","venue":"arXiv (Cornell University)","work_id":"66f5f754-3e9b-4937-8c02-50d2a6be81d1","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2412.08821","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:90ec010d82d90e2c95971e986bcfcab1f5026459f86ee97b10e9cefa2dcb7269","observation_id":"4282b008-a885-4b7c-ab7e-b829f150f779","resolution":{"observed_at":"2026-05-12T06:36:28.938338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"o ppel, Markus Spanring, Andreas Auer, Oleksandra Prudnikova, Michael K Kopp, G \\","venue":null,"work_id":"1a55c6e7-8751-4a28-9c91-bee951ff324f","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:9f9afe24fbb4c835f059eb1605993937e5656b1e64dc5e2af325ffac2df03b68","observation_id":"17154d6a-e643-4fcb-8200-4c6b80ffbb30","resolution":{"observed_at":"2026-05-12T16:46:42.469235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.06297","last_updated":"2016-01-07T22:41:10Z","snapshot_observed_at":"2026-08-04T23:18:10.428920Z","submitted_at":"2015-11-19T18:40:22Z","title":"Conditional Computation in Neural Networks for faster models","version":2},"cited_work":{"arxiv_id":"1511.06297","doi":null,"metadata_source":"pith","pith_arxiv_id":"1511.06297","snapshot_observed_at":"2026-07-08T11:44:51.084480Z","title":"Conditional Computation in Neural Networks for faster models","venue":"cs.LG","work_id":"55b55714-5d1e-4aa7-ada8-2f5ad0a80e39","year":2015},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1511.06297","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:62c1e0525a1cb1e662ef8f56e1ebf7a40df6036606dac7bcc1523bfba4082dbb","observation_id":"949ad529-ec5c-48ed-a348-2fbcaac31232","resolution":{"observed_at":"2026-05-12T06:41:24.655364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A neural probabilistic language model","venue":null,"work_id":"d44afd0e-c868-4787-987e-6806d8b00b2d","year":2003},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:7d4fa76a3c8b566ba05862d3e0e7c35ebca4f9e11c6334bbef83d08679e20dc2","observation_id":"0bab88ea-4fe3-4761-9724-65e1cded232b","resolution":{"observed_at":"2026-05-12T16:46:42.293983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Piqa: Reasoning about physical commonsense in natural language","venue":null,"work_id":"d7e885e1-813b-4cfb-9b5d-8d6d7d772c33","year":2020},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:620f8024dcc2030e301a90ee1e6d1c489df77b05e6882a6e6b782f5674989ea7","observation_id":"fdae45e8-ae3d-465c-b256-6f2d42879c7d","resolution":{"observed_at":"2026-05-12T16:46:42.307227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are few-shot learners","venue":null,"work_id":"11d04a1d-6156-4c45-b874-087617d1982c","year":2020},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:c488ed6849ac4448c8499393e5e6bde21eb688443be4c4c8ec804a2dfe7db985","observation_id":"6a01556e-eff6-4dfc-8640-16dcbe040bbe","resolution":{"observed_at":"2026-05-12T16:46:42.389613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lee, Deming Chen, and Tri Dao","venue":null,"work_id":"7546d3a2-ea9a-4fa2-b963-74764052b835","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:48fa67051a89042908c9b1107fb6654e3fdd6c7cad9ed5774b3a9b996768c7f8","observation_id":"ae5aef9f-eea5-4b88-acf7-747a27194c85","resolution":{"observed_at":"2026-05-12T16:46:42.396132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:01be3def7acef6913b23509c147cee0653d7b9cd42fdbc494a5354ce07d19bab","observation_id":"2ef76be8-a813-47ca-819b-d8b1dd0aa5bf","resolution":{"observed_at":"2026-05-12T06:36:28.849357Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.10322","last_updated":"2019-08-27T16:53:59Z","snapshot_observed_at":"2026-08-04T09:34:19.963110Z","submitted_at":"2019-08-27T16:53:59Z","title":"Bridging the Gap for Tokenizer-Free Language Models","version":1},"cited_work":{"arxiv_id":"1908.10322","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1908.10322","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bridging the gap for tokenizer-free language models","venue":null,"work_id":"991498de-07d3-4e22-a749-861d6f717c30","year":1908},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1908.10322","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:a92c3a82ffd61c0067c48a9fbdaf247eb480ec412ec1e769ecae7b21ac2ffedb","observation_id":"6e6d25bf-49f9-4dfa-9e76-5b7fb45eac1b","resolution":{"observed_at":"2026-05-12T06:36:29.262266Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hierarchical multiscale recurrent neural networks","venue":null,"work_id":"b5619bad-d29b-400f-bd67-7ad215707be5","year":2017},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:e0f69011d16835f5be15b355ef83b05e5c9201fe206f1b70afbf0d7aa1b6db77","observation_id":"e9e5463b-26bd-4b2f-872b-01021a4a18c8","resolution":{"observed_at":"2026-05-12T16:46:42.310954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"B ool Q : Exploring the surprising difficulty of natural yes/no questions","venue":null,"work_id":"180089a7-6cd0-4486-87a3-93ebf58bbe98","year":2019},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:0cb94fc1a3dbe6ad92d23dbc0c17c08664eb0ce7d137a7fd76a2e5df6a7ac241","observation_id":"d9b151f9-1c12-4ae0-9a15-1b3eafd9a618","resolution":{"observed_at":"2026-05-12T16:46:42.392949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Clark, Dan Garrette, Iulia Turc, and John Wieting","venue":null,"work_id":"c90f9a89-e03c-4b6f-affb-cff07a53a676","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:4a6167dea8b44cb10306e8144fb58edf2cc615bd0c64ff1cba281c9935873502","observation_id":"b01e1819-f1a9-48f4-9c02-ada5a17b18e3","resolution":{"observed_at":"2026-05-12T16:46:42.507672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":"1803.05457","doi":"10.1162/tacl_a_00448.https://aclanthology.org/2022.tacl-1.5","metadata_source":"pith","pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","venue":"cs.AI","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","year":2018},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:4a168a8056af71b0a7872f71233566ff01063d05662c2aef12f3fbb4be119d46","observation_id":"3b7a85e4-e7d7-4162-87c2-e766bcdbc3c0","resolution":{"observed_at":"2026-05-12T06:36:29.139734Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.04672","last_updated":"2022-08-25T17:10:53Z","snapshot_observed_at":"2026-07-06T13:29:47.927628Z","submitted_at":"2022-07-11T07:33:36Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","version":3},"cited_work":{"arxiv_id":"2207.04672","doi":"10.18653/v1/w19-5207","metadata_source":"pith","pith_arxiv_id":"2207.04672","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","venue":"cs.CL","work_id":"68c8336c-d20e-40fa-ba9c-89459da6fc1a","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2207.04672","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:17ec89ecdea4da7823cd14360189cc50121ca219d89d82f87a7f85139b52ac49","observation_id":"eb5807a7-a5f9-4d7a-a453-3ad7ae5517fa","resolution":{"observed_at":"2026-05-12T17:54:14.284800Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T17:02:41.104488Z","title":"Mo EUT : Mixture-of-experts universal transformers","venue":null,"work_id":"08049bfa-6997-41a9-853b-7eabf40d1984","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:1e8bb78afd18727ec1b37d87605c9259f0367de2f1008cb5f36b65b23290a5d1","observation_id":"7ade6638-ef35-4795-83e4-6b3a6dcee219","resolution":{"observed_at":"2026-05-12T16:46:42.458156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01035","last_updated":"2024-02-07T10:51:11Z","snapshot_observed_at":"2026-07-06T17:24:03.237617Z","submitted_at":"2024-02-01T21:49:34Z","title":"Getting the most out of your tokenizer for pre-training and domain adaptation","version":2},"cited_work":{"arxiv_id":"2402.01035","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01035","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Getting the most out of your tokenizer for pre-training and domain adaptation","venue":null,"work_id":"68fd08a9-8f2c-444f-b2cb-68045b1b928e","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2402.01035","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:49980dcf5ba941a62f6d8a95e3966c91cf6d75ca5b267422e6332a421a3d4592","observation_id":"3fa4007d-2fa5-402c-a808-8700e9a1430f","resolution":{"observed_at":"2026-05-12T06:36:28.790238Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Funnel-transformer: Filtering out sequential redundancy for efficient language processing","venue":null,"work_id":"dda521ac-e5eb-4652-9dc8-233b0e389f22","year":2020},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:8b6f1e646911b9b2b4c4e965f0adafe90ed767c485ae55defbf466ea7deb2dcc","observation_id":"db0c604f-3d80-454a-89f7-4c87eb1f8c52","resolution":{"observed_at":"2026-05-12T16:46:42.298623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transformers are SSM s: Generalized models and efficient algorithms through structured state space duality","venue":null,"work_id":"f67dfd08-b814-4617-92da-a5054e12329c","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:383688f7e8dfd72dd49d266bd5586dbe232036eaac5525cb24eb5242ef9b9891","observation_id":"104d1fa3-97d4-4013-825f-852973f78e50","resolution":{"observed_at":"2026-05-12T16:46:42.442994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Universal transformers","venue":null,"work_id":"bcda2658-f0b9-4177-b3a8-bdee62a89bdd","year":2019},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:41fd46e3a59b03a52996300d04244b208bbac98b654981418a187083e222b4bf","observation_id":"9484715a-1b7d-40bb-b06d-e411caf3904a","resolution":{"observed_at":"2026-05-12T16:46:42.402438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n19-1423","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BERT : Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"Proceedings of the 2019 Conference of the North","work_id":"3e3c8ac8-b858-4b22-af32-393d98c883e0","year":2019},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:12dba9c490e615feef069ff338a11951367fc159848410b5ef77e0340b103e63","observation_id":"e0e1c019-e0d4-4742-88b2-d410cf5cdd24","resolution":{"observed_at":"2026-05-12T04:06:19.682237Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T13:38:13.85894+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T13:38:13.85894+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A new algorithm for data compression","venue":null,"work_id":"c77b9553-8985-4e8e-95f7-9ae3e3b5956e","year":1994},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:7c56ff345a9e2f6ef942e3005754f9fc483d269ef0f10e4017fbd88bc0fa6b57","observation_id":"b19ad1c0-9ce3-4c20-a8d0-282dd964c99d","resolution":{"observed_at":"2026-05-12T16:46:42.379818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14073","last_updated":"2024-02-21T19:01:03Z","snapshot_observed_at":"2026-08-06T14:06:19.638039Z","submitted_at":"2024-02-21T19:01:03Z","title":"Improving Language Understanding from Screenshots","version":1},"cited_work":{"arxiv_id":"2402.14073","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.14073","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving language understanding from screenshots","venue":null,"work_id":"7fc7735f-811e-4298-8572-86d0d14d7e22","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2402.14073","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:a473c31db6ec03df5e5de15e95fc078a49dd9a4823634c0175d1a5c50219188b","observation_id":"9138ff7f-3b7f-41bf-8e83-4b0a7ea88ed9","resolution":{"observed_at":"2026-05-12T06:36:29.128760Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05171","last_updated":"2025-02-17T17:14:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-07T18:55:02Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","version":2},"cited_work":{"arxiv_id":"2502.05171","doi":"10.1016/0041-5553(81)90075-6","metadata_source":"pith","pith_arxiv_id":"2502.05171","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","venue":"cs.LG","work_id":"1ee7474f-a930-486e-897c-207b8755f2c9","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2502.05171","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:8d631a4c5b28a935a8e595370af62b08e9cdca1cc5a599192915365af38c8095","observation_id":"3529846b-e939-45e3-a587-7da592bd7876","resolution":{"observed_at":"2026-05-12T15:39:41.804715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lee, and Dimitris Papailiopoulos","venue":null,"work_id":"56dcb877-2964-45c9-b672-3c88e21cf8a4","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:e5486109b303942e0ce3e49b9f5157f3d09855400029627f6f6bb219c5da86e8","observation_id":"8741a2a4-54c7-4e18-a16b-c2529ea277f1","resolution":{"observed_at":"2026-05-12T16:46:42.484868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19737","last_updated":"2024-04-30T17:33:57Z","snapshot_observed_at":"2026-08-03T23:22:09.849239Z","submitted_at":"2024-04-30T17:33:57Z","title":"Better & Faster Large Language Models via Multi-token Prediction","version":1},"cited_work":{"arxiv_id":"2404.19737","doi":"10.48550/arxiv.2404.19737","metadata_source":"pith","pith_arxiv_id":"2404.19737","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Better & Faster Large Language Models via Multi-token Prediction","venue":"cs.CL","work_id":"7235774b-df35-4d85-a7ef-7deebd473172","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2404.19737","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:6a852ec992bfad2420fd05e3bd1dbab7f3511d7679944d15b9d2e365b726ab61","observation_id":"a3ebf462-cd4d-47b4-ad71-f1d60bfd6ba2","resolution":{"observed_at":"2026-05-16T12:26:09.884823Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MANT a: Efficient gradient-based tokenization for end-to-end robust language modeling","venue":null,"work_id":"f2921226-415d-492d-97e9-660d11df5512","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:492649973220b36e3ac8f69f546da7036d05a3ab4765ec8fc894837cc81f529a","observation_id":"533c8037-8125-47ae-bd69-a95e6ce28393","resolution":{"observed_at":"2026-05-12T16:46:42.461769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:21caae11d38dc5d7358c00f7b01418339e32acffce7bac6848667f602f5fba87","observation_id":"9a10a360-d88d-46e3-b34f-e83cc777fd41","resolution":{"observed_at":"2026-05-12T06:36:29.003298Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T17:02:41.062194Z","title":"Think before you speak: Training language models with pause tokens","venue":null,"work_id":"888fbea8-b1c5-4725-8433-050b8838392a","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:06e79b4b043f9b172e446b23c2d215c65d62ce979afdeba8f7985be2f40dc968","observation_id":"1d1635a5-75c3-4166-8d12-393f709b1fec","resolution":{"observed_at":"2026-05-12T16:46:42.376239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1308.0850","last_updated":"2014-06-05T16:04:02Z","snapshot_observed_at":"2026-08-05T10:16:44.137827Z","submitted_at":"2013-08-04T21:04:36Z","title":"Generating Sequences With Recurrent Neural Networks","version":5},"cited_work":{"arxiv_id":"1308.0850","doi":"10.48550/arxiv.1308.0850","metadata_source":"pith","pith_arxiv_id":"1308.0850","snapshot_observed_at":"2026-07-10T20:07:33.400866Z","title":"Generating Sequences With Recurrent Neural Networks","venue":"cs.NE","work_id":"3da63ce3-c701-48e0-a32e-756f2712b58c","year":2013},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1308.0850","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:60ba27f700686ea938f02426bb9724f5189111bb7c4ea39ba74f19bb4cfa89f8","observation_id":"448c6cfa-953e-4975-a68c-a7094cba5aa4","resolution":{"observed_at":"2026-05-12T06:36:28.887084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1603.08983","last_updated":"2017-02-21T16:21:21Z","snapshot_observed_at":"2026-08-04T14:24:45.814840Z","submitted_at":"2016-03-29T22:09:00Z","title":"Adaptive Computation Time for Recurrent Neural Networks","version":6},"cited_work":{"arxiv_id":"1603.08983","doi":"10.48550/arxiv.1603.08983","metadata_source":"pith","pith_arxiv_id":"1603.08983","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Adaptive Computation Time for Recurrent Neural Networks","venue":"cs.NE","work_id":"75565443-173e-479c-b0e7-d2464e7630be","year":2016},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1603.08983","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:bf67b981f37b4f9edda8a8490d928e245c234e292ca40411c982ffa6b5c75afc","observation_id":"a3b8d8a6-1cfa-49c3-8122-20fda793484b","resolution":{"observed_at":"2026-05-12T11:53:22.734695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fast and expressive multi-token prediction with probabilistic circuits","venue":null,"work_id":"dc3b648f-f3c5-4dcb-beba-86df609d0039","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:1cc9a35e027717743a3918dc6af8dd43bd6b82a32ecc161ddcff706e5c832e80","observation_id":"c93042f3-5792-45b5-bacd-e505f59884de","resolution":{"observed_at":"2026-05-12T16:46:42.512028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":"a589548c-3674-4236-b23c-9ffa78f6a7a3","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:be093afd1cce2d3e66e2ebc5691cfa05cb751ff03fa92fa95ba36d1def2f0bbc","observation_id":"df9c3301-d60b-4170-9348-944cce3b88c7","resolution":{"observed_at":"2026-05-12T16:46:42.434981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08446","last_updated":"2025-02-11T18:59:26Z","snapshot_observed_at":"2026-08-06T16:31:19.141189Z","submitted_at":"2024-06-12T17:37:09Z","title":"OLMES: A Standard for Language Model Evaluations","version":2},"cited_work":{"arxiv_id":"2406.08446","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08446","snapshot_observed_at":"2026-07-03T16:48:39.417321Z","title":"Olmes: A standard for language model evaluations","venue":null,"work_id":"5eb8c12c-3fc3-4566-b5e8-a7408678ffed","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2406.08446","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:04aa2cc29b11e8b45423003c5f9adf0cd8d223ef30ee6c9355c3f5bcab4d1b05","observation_id":"e179736f-a2fd-408d-afde-1ea02f97e1af","resolution":{"observed_at":"2026-05-12T06:36:29.120033Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.06351","last_updated":"2026-05-07T16:40:07Z","snapshot_observed_at":"2026-07-06T22:48:08.821342Z","submitted_at":"2026-03-06T14:59:11Z","title":"DC-DiT: Adaptive Compute and Elastic Inference for Visual Generation via Dynamic Chunking","version":2},"cited_work":{"arxiv_id":"2603.06351","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.06351","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DC-DiT: Adaptive Compute and Elastic Inference for Visual Generation via Dynamic Chunking","venue":"cs.CV","work_id":"d87b1412-41e2-4c6a-af03-1173195db729","year":2026},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2603.06351","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:26a1321129912c3b17fd99616d700d4c7bc4f4daefe34fea5fc55c8bd5a28e06","observation_id":"52d0423a-5349-463c-b03d-a209ee79f5e3","resolution":{"observed_at":"2026-05-12T06:36:29.240841Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"General-purpose, long-context autoregressive modeling with perceiver AR","venue":null,"work_id":"d088d0ed-931d-4a2b-ad67-5ad8be8c47be","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:86566e31e1f30d88119d3df997ee7671ba5abbc15c03ac61c0d97902eb0b154d","observation_id":"bd8a292b-ab5b-435c-970d-6f08198a24d0","resolution":{"observed_at":"2026-05-12T16:46:42.428019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:37:20.675521Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":"42c395e2-9c11-4300-9b02-bf1da5b5606b","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:24239abc03ea695bd62080ce8ef7204fc5ea995f15f9937d31a68106e63f1649","observation_id":"0d9e771a-d6fa-4a48-88ec-1e9de7d0b470","resolution":{"observed_at":"2026-05-12T16:46:42.420285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Block transformer: Global-to-local language modeling for fast inference","venue":null,"work_id":"21dd878e-8bfe-42f7-b776-5a8c32025a49","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:0c7aa22721498b984cec2f2a565f5b9d7a4518424d6d4537c34434bddd8215fb","observation_id":"6fb07288-5514-4f0a-ab1c-cae81b06a61c","resolution":{"observed_at":"2026-05-12T16:46:42.438923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep networks with stochastic depth","venue":null,"work_id":"e21dbad2-6307-4e45-a810-1834e9f154d4","year":2016},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:84805725955c520545e208044bad95ea46daffe2e53cd4fa046549d8bae8134b","observation_id":"763c9027-4771-4aaa-9e7a-596e99e52afd","resolution":{"observed_at":"2026-05-12T16:46:42.416367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.21420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conceptmoe: Adaptive token-to-concept compression for implicit compute allocation","venue":null,"work_id":"56206cbf-f2f0-47a5-9577-1bcf919c7643","year":2026},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d757ec9b89ba5345241fe1581ac95dfa197fddaa512c75eed810e761c804c1fb","observation_id":"29771bf6-6e85-4dd7-b0a8-9a157412cf50","resolution":{"observed_at":"2026-05-12T06:36:29.009644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Character-level language modeling with hierarchical recurrent neural networks","venue":null,"work_id":"21a43683-7234-4c9a-8646-f1a2e11cc565","year":2017},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:71d15ea61d1de81349891cf31aa3ea7c258c7f8504da7ec7ee28b6eda61c4840","observation_id":"c19a60f8-b258-42c8-85ac-9e512f693668","resolution":{"observed_at":"2026-05-12T16:46:42.281143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.07955","last_updated":"2025-07-15T09:06:11Z","snapshot_observed_at":"2026-08-06T18:25:51.582642Z","submitted_at":"2025-07-10T17:39:37Z","title":"Dynamic Chunking for End-to-End Hierarchical Sequence Modeling","version":2},"cited_work":{"arxiv_id":"2507.07955","doi":"10.48550/arxiv.2507.07955","metadata_source":"pith","pith_arxiv_id":"2507.07955","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Hwang, B","venue":"cs.LG","work_id":"2c24205d-e043-4482-800c-bb7d2d61ad6c","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2507.07955","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:a30a35f7cb9efec7897317dc1800161c7c3aac2758f35f8c72e8079f24be48f8","observation_id":"c9d4a8af-d64c-413b-9b18-497755608a41","resolution":{"observed_at":"2026-05-12T06:36:28.977825Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.14795","last_updated":"2022-03-15T22:37:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-30T17:53:34Z","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","version":3},"cited_work":{"arxiv_id":"2107.14795","doi":"10.48550/arxiv.2107.14795","metadata_source":"pith","pith_arxiv_id":"2107.14795","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","venue":"cs.LG","work_id":"92bf7a73-2bef-4de4-8957-a9233d60b416","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2107.14795","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:45cdab314c39b0abb06df4ac3dddecafc1e0444bb716878ed21bbd5d964acd29","observation_id":"2fa510aa-69a7-4e8d-bc35-d13d9db76557","resolution":{"observed_at":"2026-05-15T19:47:14.368858Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perceiver: General perception with iterative attention","venue":null,"work_id":"65c6eb00-7778-4973-aa03-d84d54718c6a","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:062678644029b1bcd11a44a4d2caf8c778700e4834a8923758ba45d98e2e6a67","observation_id":"32ec8fda-688a-48da-a1da-2688492f6782","resolution":{"observed_at":"2026-05-12T16:46:42.383706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"`` low-resource '' text classification: A parameter-free classification method with compressors","venue":null,"work_id":"48d44ef4-c732-4509-871c-287f49869bba","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:9f0a551801047030072f54b0166dc69452ab41481831f86ab2aa61d604654573","observation_id":"303385c4-0d43-4ac6-b8f8-3d9d6249d87c","resolution":{"observed_at":"2026-05-12T16:46:42.343094Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20771","last_updated":"2025-04-02T03:23:02Z","snapshot_observed_at":"2026-07-06T19:40:34.668379Z","submitted_at":"2024-10-28T06:14:12Z","title":"MrT5: Dynamic Token Merging for Efficient Byte-level Language Models","version":3},"cited_work":{"arxiv_id":"2410.20771","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.20771","snapshot_observed_at":"2026-07-04T04:39:34.638237Z","title":"Mrt5: Dynamic token merging for efficient byte-level language models.arXiv preprint arXiv:2410.20771","venue":null,"work_id":"ae1d2ecb-4725-43f0-8aa8-07aa8324ae1f","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2410.20771","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:40171e11833f62928ceede868bf3d03c9e7d431c64ecf53efcb61fb940e61f24","observation_id":"b180b5a6-dc70-4904-b5e7-e5b1062322d6","resolution":{"observed_at":"2026-05-12T06:36:28.928013Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Subword regularization: Improving neural network translation models with multiple subword candidates","venue":null,"work_id":"43b4f45f-5c63-4910-a5a6-47eaa1410546","year":2018},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:768385859828a4ba30a0f5d8f136c1172d43b5ec5077186a9b2fac995da4ebbe","observation_id":"abffccac-c31b-4d9e-902c-40b6ea87487d","resolution":{"observed_at":"2026-05-12T16:46:42.353721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"S entence P iece: A simple and language independent subword tokenizer and detokenizer for neural text processing","venue":null,"work_id":"318bc3d5-a1ba-4d71-af5d-a73b14ab52aa","year":2018},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:4155ebab2d5b7649c187ac358f8e4bcce6dd3952823762166dfb5180c3259abb","observation_id":"3d2df5aa-93d6-40e6-b645-f69dba7fb476","resolution":{"observed_at":"2026-05-12T16:46:42.285555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mamba-3: Improved sequence modeling using state space principles","venue":null,"work_id":"20968410-78ad-4a7a-bda8-99eb0f9235ac","year":2026},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d6e5e08afc1c05d0c7270b3813f25a8cbe33378e4f4dacf8d8598ccde553dbbf","observation_id":"9e563b4f-ee86-416c-bfb6-345a22a3786a","resolution":{"observed_at":"2026-05-12T16:46:42.399400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.05417","last_updated":"2024-09-27T09:03:05Z","snapshot_observed_at":"2026-07-06T18:11:48.858911Z","submitted_at":"2024-05-08T20:37:56Z","title":"Fishing for Magikarp: Automatically Detecting Under-trained Tokens in Large Language Models","version":2},"cited_work":{"arxiv_id":"2405.05417","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.05417","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2405.05417 , year=","venue":null,"work_id":"3f6c2344-5d19-4d34-bf98-0d5da7374bfd","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2405.05417","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:f5e2b72fc1577d32bfb94608191ec9d9acab169cbb3c0b87341273855c239b11","observation_id":"7861d0da-8c14-4247-a758-834f7c2ff845","resolution":{"observed_at":"2026-05-12T06:36:29.075646Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03626","last_updated":"2024-12-12T23:03:54Z","snapshot_observed_at":"2026-08-06T08:30:39.842466Z","submitted_at":"2024-04-04T17:48:28Z","title":"Training LLMs over Neurally Compressed Text","version":3},"cited_work":{"arxiv_id":"2404.03626","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03626","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lester, B., Lee, J., Alemi, A., Pennington, J., Roberts, A., Sohl-Dickstein, J., and Constant, N","venue":null,"work_id":"93265594-c964-45fd-ac73-18a3f46da9ae","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2404.03626","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:773cd3869bd39db68fd09726346128c42f47b36d4376b4ee477f637937ef6467","observation_id":"2f26fe72-8ceb-49ad-9377-9faedc558799","resolution":{"observed_at":"2026-05-12T06:36:28.952129Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"cited_work":{"arxiv_id":"2406.11794","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.11794","snapshot_observed_at":"2026-07-04T18:50:04.316326Z","title":"DataComp-LM: In search of the next generation of training sets for language models","venue":"cs.LG","work_id":"a4c88edf-049f-4a82-8f81-9110091a04ad","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2406.11794","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:1d659be7feedd0525b229ba88a4e73d2a777fdadfa6d63121bdf1c8bf35ee321","observation_id":"1d5eba93-1f43-4c78-9e57-d6d325bc3670","resolution":{"observed_at":"2026-05-17T22:58:17.761776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10691","last_updated":"2024-11-11T13:33:25Z","snapshot_observed_at":"2026-07-06T17:45:31.559370Z","submitted_at":"2024-03-15T21:21:11Z","title":"MYTE: Morphology-Driven Byte Encoding for Better and Fairer Multilingual Language Modeling","version":2},"cited_work":{"arxiv_id":"2403.10691","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10691","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InFindings of ACL 2023","venue":null,"work_id":"3b450723-b4ce-4a53-80a1-710edb292b15","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2403.10691","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:33cd2cd7ff6bf2a4fe203e586449f6db72cdbcb7ff7ca849c4bce56785bdfc5a","observation_id":"9699f24d-0927-420a-a535-75180591fe3d","resolution":{"observed_at":"2026-05-12T06:36:28.968337Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Smith, and Yejin Choi","venue":null,"work_id":"2ba7d865-b541-4d80-b5f8-f04cc3b72dfb","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:b251e43371617f7ac072ef73c795a15e524dfa10ce3215538d50da2fc6d1cf4b","observation_id":"b91d14af-756d-4463-bb37-1c242002f7aa","resolution":{"observed_at":"2026-05-12T16:46:42.277059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Decoupled weight decay regularization","venue":null,"work_id":"30ece2e7-3c79-4ec7-afc1-0cb2a9c0f7d9","year":2019},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:eb0a7dd72aa11218cfa786827570dfffb043fa96222512181dc8247ca381636d","observation_id":"29744b20-b3d1-42b1-ae19-f3cb4d3bef1f","resolution":{"observed_at":"2026-05-12T16:46:42.350401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Text rendering strategies for pixel language models","venue":null,"work_id":"e3e9d8d0-7916-4ddc-9dfc-c939290bc635","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:3c332deb2dd45acb5dad7e150e79a85e417980341fd90c6498f9919430556030","observation_id":"b20de449-b3d5-4d72-8919-2c90cc6c6d94","resolution":{"observed_at":"2026-05-12T16:46:42.405883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Starcoder 2 and the stack v2: The next generation","venue":null,"work_id":"8e565750-a879-4148-abc5-c21997e0c0d6","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:9e0a06ace8fd586473f109f9127f34ef46d9072983e177d764befef406e9e6ea","observation_id":"d928f038-711e-4ed1-b0cd-7249749f9f33","resolution":{"observed_at":"2026-05-12T16:46:42.500264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The art of prompt design: Prompt boundaries and token healing","venue":null,"work_id":"03cf320c-a66b-45e2-a12f-78af549208b1","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:b23ace4a12bb5aaecef23bd4f553ff590ff1e1a855131a60adcf5daf96888b8b","observation_id":"4ecb3e0d-da08-4a44-86df-edf53afc970e","resolution":{"observed_at":"2026-05-12T16:46:42.503996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidance","venue":null,"work_id":"cd880313-59fc-493a-88dc-0b538f973ec3","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:4ee519cc70df0f33411151e70a5e71d6845dd317597d0b57e1b113333e55116f","observation_id":"319fe42b-e335-4e72-b036-6550ee642834","resolution":{"observed_at":"2026-05-12T16:46:42.496641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering","venue":null,"work_id":"6b253942-8cdb-47e0-a922-48dc9bb54008","year":2018},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:ec1fb1202bd40f654ba22bbd7afefd683f4d0c7b72f768e6f063afc5dde68b97","observation_id":"323195a1-9d0d-4ac9-9322-5e67180d5615","resolution":{"observed_at":"2026-05-12T16:46:42.328161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.15586","doi":"10.48550/arxiv.2512.15586","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Minixhofer, T","venue":"Open MIND","work_id":"e2bb8878-7691-4bc4-9570-ece4c71261b4","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:beef9b75ac6e2900203bc9640a819af0aad4e8b3a067391589712724871ca55e","observation_id":"b6b19845-325a-49f3-846d-2831fa844e7b","resolution":{"observed_at":"2026-05-12T06:36:29.206351Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hierarchical transformers are more efficient language models","venue":null,"work_id":"15957efb-bdbe-45f4-aa2b-b0dd21d09d6b","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d0dc927870b020246492c9cc2865be5f9f65ae4502be0ab5b23bfdbb6d167be3","observation_id":"f2e5a378-ef06-4093-8ff7-0a407bcee020","resolution":{"observed_at":"2026-05-12T16:46:42.450872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient transformers with dynamic token pooling","venue":null,"work_id":"e0d9f545-2f7a-4ef7-a2f1-6ee38d0cc8c4","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:91fca2f1b00e1aa84c20f087a0fb3bf5d9c80042dafdd9e26ff2e110bfe34c31","observation_id":"6c428cfe-862d-4929-a668-421fed0f9a22","resolution":{"observed_at":"2026-05-12T16:46:42.315000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.10322","last_updated":"2025-01-20T09:33:21Z","snapshot_observed_at":"2026-07-06T20:22:32.172373Z","submitted_at":"2025-01-17T17:51:53Z","title":"Hierarchical Autoregressive Transformers: Combining Byte- and Word-Level Processing for Robust, Adaptable Language Models","version":2},"cited_work":{"arxiv_id":"2501.10322","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.10322","snapshot_observed_at":"2026-07-11T03:47:48.559145Z","title":"Hierarchical autoregressive transformers: Combining byte-\\ and word-level processing for robust, adaptable language models","venue":"cs.CL","work_id":"f1b77678-4225-4a48-aa56-d3168260cf0d","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2501.10322","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:6f5fbd3a26c8f2e3e0039a2160c2e39d07585c1a8775bf5348489b83c8a3a5ce","observation_id":"e2b73c34-d868-473f-b1c9-35637e98a049","resolution":{"observed_at":"2026-05-12T06:36:28.879364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d0ea0c1a4cefeaea3ca33e44dce765c8759bd118d2e9971be6eb20d4147d65fe","observation_id":"2a0d88a2-9ad4-4115-af90-bdca6c3c9326","resolution":{"observed_at":"2026-05-12T06:36:28.914004Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.12720","last_updated":"2026-05-13T15:04:08Z","snapshot_observed_at":"2026-08-03T04:43:10.143472Z","submitted_at":"2025-07-17T01:55:41Z","title":"FLEXITOKENS: Flexible Tokenization for Evolving Language Models","version":4},"cited_work":{"arxiv_id":"2507.12720","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12720","snapshot_observed_at":"2026-07-04T04:39:34.634847Z","title":"Flexitokens: Flexible tokenization for evolving language models","venue":"cs.CL","work_id":"4ea3b0fc-1312-4e9b-8eaa-b90f626566fa","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2507.12720","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:43a1b4ff6f5c8c2abd8e83c14c716dfaf555a54d1fea5460971fbf59225d764f","observation_id":"cd7fb21d-0d92-40c9-b092-663f8dacf775","resolution":{"observed_at":"2026-05-14T02:56:13.953344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.09871","last_updated":"2024-12-13T05:33:32Z","snapshot_observed_at":"2026-08-06T17:12:17.931581Z","submitted_at":"2024-12-13T05:33:32Z","title":"Byte Latent Transformer: Patches Scale Better Than Tokens","version":1},"cited_work":{"arxiv_id":"2412.09871","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.09871","snapshot_observed_at":"2026-07-11T03:47:48.461365Z","title":"arXiv preprint arXiv:2412.09871 , year=","venue":"cs.CL","work_id":"e1462084-176b-4aab-aa7e-ec8788bdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2412.09871","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:f6a7d2a9529e93b7f2b5a638d09f223d7da6506119d677748b601b901e1abce3","observation_id":"53a577bf-b816-4a10-ad63-442b74c0309b","resolution":{"observed_at":"2026-05-12T06:36:29.184901Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openwebmath: An open dataset of high-quality mathematical web text","venue":null,"work_id":"551e2408-3a45-445a-b7f6-dbc3cac7b04c","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:c928c49f6c2bac717eafda3e938f34ce1b7f7f5454b786884195a5aef76bb53e","observation_id":"626da5f5-2f83-4268-93c0-77ac5a6a2eaa","resolution":{"observed_at":"2026-05-12T16:46:42.323155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.24617","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dynamic large concept models: Latent reasoning in an adaptive semantic space","venue":null,"work_id":"f1b58f22-de60-4f3f-97ad-54303c22f844","year":2026},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:25fc2b98d05d339bc25a1a46db86353cde4f9ead57d269de0daf51bac3693379","observation_id":"5d331cc4-1b11-4e5f-be0f-46429cf8d53c","resolution":{"observed_at":"2026-05-12T06:36:28.796718Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.01444","last_updated":"2017-04-06T09:48:20Z","snapshot_observed_at":"2026-07-06T05:36:39.612294Z","submitted_at":"2017-04-05T14:20:28Z","title":"Learning to Generate Reviews and Discovering Sentiment","version":2},"cited_work":{"arxiv_id":"1704.01444","doi":"10.48550/arxiv.1704.01444","metadata_source":"pith","pith_arxiv_id":"1704.01444","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning to Generate Reviews and Discovering Sentiment","venue":"cs.LG","work_id":"3a589728-009a-467e-a1d9-ed3624e82dcb","year":2017},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1704.01444","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:4297891a6ac397b81f8a20e151e5298b2578e83fd4384ff6221d8adb1a87d9b4","observation_id":"ea47df25-b79d-4ca1-9240-1bf4cbffcfb2","resolution":{"observed_at":"2026-05-12T06:36:29.039412Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"77f70b47-1236-4fbb-bed7-c80cb9eda994","year":2020},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d35f17b7353ff1efaf5001abbbd2305c6c91ae94dc4cb63abdfe7c22adf3dabb","observation_id":"73f206ef-6d41-4de0-adb0-ac5abc801272","resolution":{"observed_at":"2026-05-12T16:46:42.412942Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02258","last_updated":"2024-04-02T19:28:11Z","snapshot_observed_at":"2026-07-06T17:54:47.689340Z","submitted_at":"2024-04-02T19:28:11Z","title":"Mixture-of-Depths: Dynamically allocating compute in transformer-based language models","version":1},"cited_work":{"arxiv_id":"2404.02258","doi":"10.48550/arxiv.2404.02258","metadata_source":"pith","pith_arxiv_id":"2404.02258","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mixture-of-Depths: Dynamically allocating compute in transformer-based language models","venue":"cs.LG","work_id":"6dcdd707-a37b-49a9-9531-4fc6f5b9ed84","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2404.02258","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:ea1432bb394df01ccf4b362b3c4598b1b6a5cff7cef7a69f61c946e334fcd89f","observation_id":"04407f9a-01d5-44cf-9868-0dd310ee8d78","resolution":{"observed_at":"2026-05-17T02:18:02.305925Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Solidgoldmagikarp (plus, prompt generation)","venue":null,"work_id":"d8f73359-3e0b-4520-a6d5-ca76ed3cc62e","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:92329c0963640ebd3737436ca2ee776285339c95a9e26826294982cbfda9e2e6","observation_id":"a776c116-46c5-412f-a994-dc32c9c80966","resolution":{"observed_at":"2026-05-12T16:46:42.454477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lotz, Emanuele Bugliarello, Elizabeth Salesky, Miryam de Lhoneux, and Desmond Elliott","venue":null,"work_id":"cd76f8a9-7715-42af-b4ea-8e0fc8beb250","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:b75f99705ed4e34dd63ced36899761dd0389bc68d2b67cc74ac11e07c26c00ec","observation_id":"e7f05fa1-038f-42ef-82d5-bf77bda3a08f","resolution":{"observed_at":"2026-05-12T16:46:42.465256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Winogrande: An adversarial winograd schema challenge at scale","venue":null,"work_id":"db9bf641-41cf-4b2b-859c-9a9d88562ea7","year":2020},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:7cf72dcb69f9b5bcaf6c882a695f6a0403d147afb0ff339eded94992b2aeb189","observation_id":"75c4f69a-aa95-4c86-be27-ff4bfa0b4f73","resolution":{"observed_at":"2026-05-12T16:46:42.368815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Robust open-vocabulary translation from visual text representations","venue":null,"work_id":"c9487d94-cab1-4c1a-8e27-20de127f10af","year":2021},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:366ea343beb707fb94753ad53d3c470571b5b16864a876fc67deac07d8b92d88","observation_id":"9c91c57b-26cb-46b0-b7b0-7aea5313010c","resolution":{"observed_at":"2026-05-12T16:46:42.357496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Japanese and korean voice search","venue":null,"work_id":"0b52fd2f-e54a-40b9-8100-6bff60c0aa73","year":2012},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:07f01b35233e7acb312947bb8c435872165da450f74788a8421c212d5ecb3f43","observation_id":"1358df84-fcff-433a-8038-2887c466d418","resolution":{"observed_at":"2026-05-12T16:46:42.339496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Neural machine translation of rare words with subword units","venue":null,"work_id":"9e0c983e-4c63-4519-9106-1ae40b123d32","year":2016},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:0a8f97a852c2eea9e63850b970cab198c0e55bbe79063664eb3c211b45934d8b","observation_id":"0cae3c0e-7e90-4dea-9d85-9c8524f753b2","resolution":{"observed_at":"2026-05-12T16:46:42.424027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14408","last_updated":"2024-10-06T02:17:26Z","snapshot_observed_at":"2026-07-06T18:03:55.981890Z","submitted_at":"2024-04-22T17:59:29Z","title":"SpaceByte: Towards Deleting Tokenization from Large Language Modeling","version":3},"cited_work":{"arxiv_id":"2404.14408","doi":"10.48550/arxiv.2404.14408","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.14408","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"13 Proxy Compression for Language Modeling Schuster, M","venue":"arXiv (Cornell University)","work_id":"715a6308-8e7f-4a22-b536-7695e6432ff0","year":2012},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2404.14408","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:8d698bbd832116895ef6c8a8f7ff7b2923feacf3568d48847478cce29eb182da","observation_id":"9b5ee4b2-cd9a-4ea0-b7ab-431894f91dc0","resolution":{"observed_at":"2026-05-12T06:36:29.090174Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blockwise parallel decoding for deep autoregressive models","venue":null,"work_id":"b30690b7-2655-4b46-b6be-5794ca394d04","year":2018},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:b5562702e8498545f112b38fc8869aa7f94ea64a6a4d8014d0697bcdd5e2e13b","observation_id":"cb630b05-c1e7-4b16-8bd7-fa16816d1380","resolution":{"observed_at":"2026-05-12T16:46:42.290011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generating text with recurrent neural networks","venue":null,"work_id":"4e1b70f2-00e7-4fc4-955e-a0eb61ef55b0","year":2011},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:60b8000551095d50e604e22032685e7af0f81c7a7bf1f13c12b341c2d692af57","observation_id":"4662c9dc-dd95-470e-bfde-e5aed21e6f98","resolution":{"observed_at":"2026-05-12T16:46:42.361809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sparse universal transformer","venue":null,"work_id":"bbadab72-bd93-4bc0-af89-2edff07dba39","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d517d4d05f695061d30faf67e970dc84290768cbb5f9958b762b6acadd213b9f","observation_id":"818ac257-47a3-4845-b669-46133384fa50","resolution":{"observed_at":"2026-05-12T16:46:42.431444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tran, Sebastian Ruder, Jai Gupta, Hyung Won Chung, Dara Bahri, Zhen Qin, Simon Baumgartner, Cong Yu, and Donald Metzler","venue":null,"work_id":"30890272-47d9-423b-9e69-4085870d9a52","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:09a7ee5a837427692218d2e501369c1b290fcf9113a6a077f7929f34bc444cc6","observation_id":"3fd762a8-b3b8-45ab-9edf-0c209165ee4d","resolution":{"observed_at":"2026-05-12T16:46:42.365289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14761","last_updated":"2025-06-17T17:55:11Z","snapshot_observed_at":"2026-08-05T17:25:40.150600Z","submitted_at":"2025-06-17T17:55:11Z","title":"From Bytes to Ideas: Language Modeling with Autoregressive U-Nets","version":1},"cited_work":{"arxiv_id":"2506.14761","doi":"10.48550/arxiv.2506.14761","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14761","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Videau, M., Idrissi, B","venue":"ArXiv.org","work_id":"c3f54890-0f12-4f37-ae35-986186e6e231","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2506.14761","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:1508882af5181ffce3e9cb3db97c86790a71fd1a2da40de17f0355382ef55144","observation_id":"b906c358-94d1-46ea-9b78-cf7d5c7a9722","resolution":{"observed_at":"2026-05-12T06:36:29.198403Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13660","last_updated":"2024-08-09T20:18:57Z","snapshot_observed_at":"2026-07-06T17:20:03.996763Z","submitted_at":"2024-01-24T18:53:53Z","title":"MambaByte: Token-free Selective State Space Model","version":3},"cited_work":{"arxiv_id":"2401.13660","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.13660","snapshot_observed_at":"2026-07-03T12:48:11.761335Z","title":"N., and Rush, A","venue":null,"work_id":"7408d5c4-b707-4822-a8dd-a52f418b53d6","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2401.13660","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:5d1d2636bb1b7ce9581b1baa9c7010a1c2d7bfbc6858e72c12c45bdf7a2b8c91","observation_id":"eb950473-1012-45be-9eac-ab678a94bd2e","resolution":{"observed_at":"2026-05-12T06:36:29.110367Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.24824","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T21:26:14.390218Z","title":"Parallel loop transformer for efficient test-time computation scaling.CoRR, abs/2510.24824","venue":null,"work_id":"7099cd19-819b-49b0-98ea-e7d1789e44b9","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:e96b0f99507123c08d1ddda01b33fd47fe0e844a3616731daa4b7b3c983245a6","observation_id":"1aed69f3-a39b-4bf3-9295-2f7d1846d93f","resolution":{"observed_at":"2026-05-12T06:36:29.026786Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14992","last_updated":"2025-04-24T04:13:49Z","snapshot_observed_at":"2026-07-06T21:12:28.608235Z","submitted_at":"2025-04-21T09:41:26Z","title":"Efficient Pretraining Length Scaling","version":2},"cited_work":{"arxiv_id":"2504.14992","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.14992","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient pretraining length scaling","venue":null,"work_id":"c76b2922-66dd-4765-bebc-a1156d708d6c","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2504.14992","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:e4e3397dec1de6bd98aee43ce556f911726783e107db1ffbb19133b8b42a8b08","observation_id":"f683faf2-3189-4eec-9029-366a612255d5","resolution":{"observed_at":"2026-05-12T06:36:29.285778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1609.08144","last_updated":"2016-10-08T19:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-09-26T19:59:55Z","title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","version":2},"cited_work":{"arxiv_id":"1609.08144","doi":"10.18653/v1/2021.eacl-main.163","metadata_source":"pith","pith_arxiv_id":"1609.08144","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","venue":"cs.CL","work_id":"e294e5a1-5dd2-44a0-b348-adbd62fe1916","year":2016},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/1609.08144","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:4222bcb599cfae18a8010903bca3c6463f87c4d04e692c3ee8ea8a576bcd2d74","observation_id":"6c881321-048e-4011-9d33-d52fbfe35ff7","resolution":{"observed_at":"2026-05-12T15:21:29.028955Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"B y T 5: Towards a token-free future with pre-trained byte-to-byte models","venue":null,"work_id":"db2dc9b8-a116-4c8b-ae90-4a1c88f88daa","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:6b4f536e7c243abdd3c6616402854c9083694c3090c97a627bf1f149cf13ffd7","observation_id":"fc9d9a3b-a53c-445e-a988-fe3fd882444f","resolution":{"observed_at":"2026-05-12T16:46:42.302910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11214","last_updated":"2024-11-14T03:53:56Z","snapshot_observed_at":"2026-07-06T18:31:57.866202Z","submitted_at":"2024-06-17T05:13:25Z","title":"Problematic Tokens: Tokenizer Bias in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.11214","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11214","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Problematic tokens: Tokenizer bias in large language models","venue":null,"work_id":"9cf09fd5-75c5-4838-a9ae-1d6f850423a0","year":2024},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"cited_paper":"/paper/2406.11214","citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:dec466e1f459f870f71aea6174042af137c16d053194ebbcdd67a651a52e394e","observation_id":"daccf9ec-9873-4498-94a3-b761c17004b8","resolution":{"observed_at":"2026-05-12T06:36:29.172693Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling embedding layers in language models","venue":null,"work_id":"69e2b752-2e3d-438a-bde7-8bea1603ba11","year":2025},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:00d9955d9d4ef517be934512df1aa93877a2c3b76302f53b9d921d9033f8d83c","observation_id":"77da302a-c137-4766-8f78-592c9548acb2","resolution":{"observed_at":"2026-05-12T16:46:42.493134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MEGABYTE : Predicting million-byte sequences with multiscale transformers","venue":null,"work_id":"8dd0925e-bcd4-47cc-8958-636d75439d75","year":2023},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d099d53d2db667000360da0aaa459fe2629cde2c6fe490c356e671c4419cebab","observation_id":"17cbaf07-d478-487a-9ddb-132efd34862c","resolution":{"observed_at":"2026-05-12T16:46:42.346658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"H ella S wag: Can a machine really finish your sentence? In Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics","venue":null,"work_id":"a65cf6fa-d3db-44c1-a5bc-0e3258cc2b1c","year":2019},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:2e25b8caee462c3a62ffcce2b7fc49de21989c89089ae6658f9ec99b260914c2","observation_id":"ddd6ee95-0fdc-4e84-9b82-880d6749e8ac","resolution":{"observed_at":"2026-05-12T16:46:42.335783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ponder LM : Pretraining language models to ponder in continuous space","venue":null,"work_id":"c1f00c12-d163-41e7-abc6-4c6dd7f13dbd","year":2026},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:455d81d26b7a56d1e1ec133e078f49695e3d235300a02f931c334dac5749765d","observation_id":"7d255cad-4e4f-4144-8e01-2a0a326d65e1","resolution":{"observed_at":"2026-05-12T16:46:42.481482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Linear complexity randomized self-attention mechanism","venue":null,"work_id":"c6b30b4c-52e6-4ee9-95d8-990d99539eb9","year":2022},"citing_paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-12T04:05:28.713898Z"},"links":{"citing_paper":"/paper/2605.09630"},"observation_digest":"sha256:d7974a5a4d2f87683515a00ad7a3681972dec01de2a91dd1a72f29217018f6d9","observation_id":"f24ce0fc-075b-4fac-b0fc-10ba1da542c1","resolution":{"observed_at":"2026-05-12T16:46:42.372163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.09630","last_updated":"2026-05-10T16:18:22Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-31T18:27:32.630942Z","submitted_at":"2026-05-10T16:18:22Z","title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":10,"parse_uncertain":0,"unresolved":1,"verified_exact":31,"verified_fuzzy":58},"total_outbound_references":104},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 100 of 104 outbound references and 0 inbound Pith citation observations for arXiv:2605.09630."}