{"as_of":"2026-08-08T20:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:338f50b2366d11f76435c7046a3a55c4c95e41dae1ecb3d23ff8aa26527b4692","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:50:02.022181Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:49:57.859145Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T14:50:02.310312Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"cited_work":{"arxiv_id":"2505.17446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17446","snapshot_observed_at":"2026-08-07T14:50:02.310312Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","venue":"cs.CL","work_id":"7adbccd3-3d74-4b74-bf5b-4e0a4a92d5a9","year":2025},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.859145Z"},"links":{"cited_paper":"/paper/2505.17446","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:b1141fa18e1a7ab4a64e1a309d4e2f9a807bc6e076609887bf8f467b03783783","observation_id":"8cf812e2-ced0-49ae-874a-a2004ad39b8e","resolution":{"observed_at":"2026-08-07T14:50:02.409712Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.17446/citation-record","integrity":"/paper/2505.17446/integrity","json":"/paper/2505.17446/citation-record.json","paper":"/paper/2505.17446"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"cited_work":{"arxiv_id":"2505.17446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17446","snapshot_observed_at":"2026-08-07T14:50:02.310312Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","venue":"cs.CL","work_id":"7adbccd3-3d74-4b74-bf5b-4e0a4a92d5a9","year":2025},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.859145Z"},"links":{"cited_paper":"/paper/2505.17446","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:b1141fa18e1a7ab4a64e1a309d4e2f9a807bc6e076609887bf8f467b03783783","observation_id":"8cf812e2-ced0-49ae-874a-a2004ad39b8e","resolution":{"observed_at":"2026-08-07T14:50:02.409712Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.415495Z","title":"Throughout this study, we used HuBERT [7] as an SSL model and extracted representations from the ninth layer","venue":null,"work_id":"7a3509e9-ef31-44fc-bb55-2b5cac99e785","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.887646Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:8be6b3bd238f4b6590e03b949fc430ea2dc8b05196370d1210a49fad8f28360e","observation_id":"0c1fb68c-fbc5-45de-b901-2f473c3febeb","resolution":{"observed_at":"2026-08-07T14:50:09.461214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.282625Z","title":null,"venue":null,"work_id":"08af9200-07e6-4497-b2bf-7f12dbf29b28","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.995598Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:451bcb1e0ffdcb21b43e569d003daf78cd620569b4d8df9a7c41de9bfe1cfd6f","observation_id":"139ffd34-fb6c-4203-bc97-72dd762a371d","resolution":{"observed_at":"2026-08-07T14:50:09.330352Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.156762Z","title":null,"venue":null,"work_id":"26a31615-d59d-4669-ab64-51e3d7d17e8e","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.128986Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:45379efb7f7181a552a9c37303a583327af402fa260ae9240c3172d892419e97","observation_id":"fe7a207b-24d1-47c7-aa0f-b692ededf434","resolution":{"observed_at":"2026-08-07T14:50:09.205464Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.032832Z","title":null,"venue":null,"work_id":"6dfafb07-f3c7-4068-aaae-3beb657972e5","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.238918Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:50d70a94cb2e8be1dce65339d67d8aab484c41763a27e5d7f86bb83a4ea47766","observation_id":"80c5be5a-d540-48db-a5fe-c97e88c5ee36","resolution":{"observed_at":"2026-08-07T14:50:09.076506Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.908184Z","title":"Dataset As a training set for SLM, we used LibriSpeech [17], a 960-hour English audiobook corpus","venue":null,"work_id":"d86554de-95bc-447e-9a24-dd31ce178874","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.300160Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:9806b1f1367bd282848a7f7fd484f9bd13d52addf90d3c7ebcf2582168de2490","observation_id":"6f133b16-38b6-4852-aea0-89f614884e69","resolution":{"observed_at":"2026-08-07T14:50:08.949260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.805177Z","title":"Figure 2 shows results on fixed boundary settings","venue":null,"work_id":"8068a7e2-544d-407b-ae3b-d1c8ce725be3","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.407640Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:63620780089e67d71307d03b1ba7203d670e4398abb076fab7f0f9d8473e419d","observation_id":"0845f3d5-355d-4e8a-974b-cb9c7422d12a","resolution":{"observed_at":"2026-08-07T14:50:08.846595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.667631Z","title":"yonder\"","venue":null,"work_id":"b7b7773d-e325-42c6-83b7-370a36a0328e","year":2016},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.481649Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:100a63c9cf8fdafa54356a9542bf2ee826776a883c63e93a544d63c91f03085e","observation_id":"c225dd39-0b87-4151-a99c-c7a4a2ac90da","resolution":{"observed_at":"2026-08-07T14:50:08.730832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.534929Z","title":"We conducted mul- tiple speech tokenizations based on the combination of the fixed/variable segmentation and the cluster size","venue":null,"work_id":"7da09f53-d01b-4671-a821-1cc82fe0a3fa","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.570939Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:d0b660fe00535cb8edbfefa323201594c4c1adc73470e51e8f503544bb857d94","observation_id":"564c9789-a8e0-4a74-943a-2d439fbab684","resolution":{"observed_at":"2026-08-07T14:50:08.591058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.343235Z","title":null,"venue":null,"work_id":"0fd6642e-43b0-4d1b-9b45-8b032444c708","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.640221Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:1c8239edff183724f148e53ca4a92142cce2b1caefde62d2ed03c22d7aa538b3","observation_id":"3fd089f2-4175-4795-b3fc-07373eed3187","resolution":{"observed_at":"2026-08-07T14:50:08.423077Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.107856Z","title":"On Generative Spoken Language Modeling from Raw Audio,","venue":null,"work_id":"bea2f7a1-3588-40c3-8fd6-a5f6fbdeb9c1","year":2021},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.753979Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:91a98ae515a161dceab3298a95c22785d0bd19b6face59cca8c69f2fbf870fb2","observation_id":"87f1e936-e923-42bf-a50c-0b65c1b67fcc","resolution":{"observed_at":"2026-08-07T14:50:08.193845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.966936Z","title":"Textually Pretrained Speech Language Models,","venue":null,"work_id":"ca87f40a-afc0-4eb5-bb6f-bbfb4f503ef7","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.883691Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:de697bd2ed3b75fd6fe10bcd1dc6cf793e25904d1a3957e8d3a13612c7bea77e","observation_id":"70af3af0-d481-4544-949c-607c1f9a2449","resolution":{"observed_at":"2026-08-07T14:50:08.028899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.804572Z","title":"Audiolm: A language modeling approach to audio generation,","venue":null,"work_id":"50356337-eb6f-475b-8e82-d09ecb628bca","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.963630Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:9b79d916531abfd91cea3ce52bc9250b7fd3c6da158868004f0d500fe8df11a4","observation_id":"5a3393b4-1624-43bc-8954-f4f254b0e0ba","resolution":{"observed_at":"2026-08-07T14:50:07.884237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.660802Z","title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model,","venue":null,"work_id":"a1dc5b7b-56f1-49fc-9c81-a98e352faeb3","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.048288Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:18ca07c170d902a7820d8adee4c04a3c809a031c45922917da61beea56fb31ec","observation_id":"2a8b485e-ab94-43da-9934-9c0545480a8d","resolution":{"observed_at":"2026-08-07T14:50:07.713715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.484840Z","title":"Representation Learning with Contrastive Predictive Coding,","venue":null,"work_id":"5ce174db-a8af-447a-bfbc-63a5d2f45e89","year":2019},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.162114Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:6ddad526595dc1dae1489fd5da8bb69011969770b71209a2bb47e75252089cb8","observation_id":"3d90e8be-6ab2-4752-bf34-49cb88ce772e","resolution":{"observed_at":"2026-08-07T14:50:07.558127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.288707Z","title":"Wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Represen- tations,","venue":null,"work_id":"9c43d405-fa6d-4b6e-8d56-b47ba4962a83","year":2020},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.229667Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:1209edad560208d600c0d607add976feb26466df5da6ed118cbdac51fc1d8d00","observation_id":"399a0d2d-8807-4c88-86db-a7dbe180c12f","resolution":{"observed_at":"2026-08-07T14:50:07.379926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.047814Z","title":"HuBERT: Self-Supervised Speech Rep- resentation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":"b8f3ecca-e9be-4113-9910-35a5df53847d","year":2021},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.368364Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:72d4bde14c51bd3735099071b107dc186fbea95ca93ea888935b6d271d9b6cd7","observation_id":"a2b91708-4c6d-450d-be67-836a8f2da1a2","resolution":{"observed_at":"2026-08-07T14:50:07.145029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.820744Z","title":"The Zero Resource Speech Benchmark 2021: Metrics and baselines for unsupervised spoken language modeling,","venue":null,"work_id":"5ad3ec76-0e39-4a2d-a876-8337a4483bec","year":2021},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.502483Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:42c7b3c4a84feffb1387dda3af1a3077570716b3f8a77bc1e0e5ac227159e90b","observation_id":"5cb4db3e-8525-4a82-9f68-88280f53a06e","resolution":{"observed_at":"2026-08-07T14:50:06.928005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.595210Z","title":"Generative Spoken Dialogue Language Model- ing,","venue":null,"work_id":"2fbf0db7-f464-4827-a08f-7070a0ac1a7b","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.637090Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:f236ad663141e8dafa2645688130f0c4d8ddafcc19ffda6538aa4eb12e15b74f","observation_id":"1bc6e752-38ab-46c8-b9ff-bbc5bdd885e5","resolution":{"observed_at":"2026-08-07T14:50:06.699670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.289082Z","title":"Direct Speech- to-Speech Translation With Discrete Units,","venue":null,"work_id":"07f189c4-47fb-4b0a-839e-c77fd8335c22","year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.792124Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:941180c5b57d4300f139f1e74bec624eab3115d5bdaa17db77b7f64d5acc5fde","observation_id":"0d1665fd-9f77-4ee3-8638-31813885e252","resolution":{"observed_at":"2026-08-07T14:50:06.423491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.036820Z","title":"Text-free prosody-aware generative spoken language modeling,","venue":null,"work_id":"96d9d5f1-ef14-4ce6-8e19-1af347896cb7","year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.930171Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:a51262f05b6e8d20134b053d7094f780052222544220cc3cb6f50a0088753a81","observation_id":"1c98c5ae-2af5-432e-a24b-6fe9e72d7ae4","resolution":{"observed_at":"2026-08-07T14:50:06.151214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.807236Z","title":"Attention is all you need,","venue":null,"work_id":"ae879658-0dd5-4230-bab2-8c8c1694a267","year":2017},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.108485Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:163a3309e8375efe8c7a03f590f2504895414b44cb9839aaf9b39e1c16a93362","observation_id":"00f4e789-1c8c-404f-ace8-d8e74dbf0b7e","resolution":{"observed_at":"2026-08-07T14:50:05.922539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.576279Z","title":"Self-Supervised Speech Representations are More Phonetic than Semantic,","venue":null,"work_id":"67c52bea-ad3f-4984-9de1-fa2ac894ec84","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.225350Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:73293032a6b4f7efedcbb8fd788ea4c2f219a2716a01aa82e9b8f0a428569a0f","observation_id":"2c6830ad-de9f-453a-9575-21f2fab6f352","resolution":{"observed_at":"2026-08-07T14:50:05.697064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.296729Z","title":"Generative Spoken Language Model based on continuous word-sized audio tokens,","venue":null,"work_id":"53d2091e-b91d-47c7-bd88-84ef4383362f","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.329313Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:324521380ae76a01028b483996193e98b2dd468d809cdd3557738bcd82f72c15","observation_id":"3639a5ac-d4ab-4078-8b27-57f965f6b5a0","resolution":{"observed_at":"2026-08-07T14:50:05.437143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.04029","last_updated":"2024-10-05T04:29:55Z","snapshot_observed_at":"2026-07-06T19:28:18.775189Z","submitted_at":"2024-10-05T04:29:55Z","title":"SyllableLM: Learning Coarse Semantic Units for Speech Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04029","snapshot_observed_at":"2026-08-07T14:50:00.432842Z","title":"SyllableLM: Learn- ing Coarse Semantic Units for Speech Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.432842Z"},"links":{"cited_paper":"/paper/2410.04029","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ef4874f9cd61b6422b486eef8bbe1ffbc095f4f3350be114b682cb1676bf1087","observation_id":"45e2ab50-0b6a-4966-8c70-02191b2dff11","resolution":{"observed_at":"2026-08-07T14:50:00.432842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.030450Z","title":"Sylber: Syllabic Embedding Repre- sentation of Speech from Raw Audio,","venue":null,"work_id":"93ec52b2-da31-4fb9-a5af-04db88ad280a","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.556666Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ede7585412d0df5a4097e3ee1d6eb75cd300a32f8f389fe1608af4f4079b4cb3","observation_id":"4367e883-40a3-4b54-b078-4ae4544f7a68","resolution":{"observed_at":"2026-08-07T14:50:05.138108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:04.750264Z","title":"Lib- rispeech: An ASR corpus based on public domain audio books,","venue":null,"work_id":"16e04614-c90d-494b-b8ec-29ce3782eab1","year":2015},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.664857Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:2ad77e746758518645b3158d0147fb246ed31c322c96566252189a4d223d8db2","observation_id":"4635ecbd-ac80-4160-9225-3d93d1ffc5be","resolution":{"observed_at":"2026-08-07T14:50:04.894664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:04.476291Z","title":"Libri-light: A benchmark for ASR with limited or no super- vision,","venue":null,"work_id":"185abeff-2ec3-4725-8b60-aa73af36e0bb","year":2020},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.802807Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ca36999c00aab48d347db9e994b3351455a969f7952e4e182da1156f73eace6c","observation_id":"3efbaa60-489f-409b-80d8-cfd65dd5e073","resolution":{"observed_at":"2026-08-07T14:50:04.615415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-07T14:50:00.957373Z","title":"OPT: open pre-trained transformer language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.957373Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:bffd78108734abeb7cbd65d7ffc1f015d00e713e77f6858ab9ab789bf97e0ded","observation_id":"1e453aac-68e8-44d7-99a5-ce410e69dde7","resolution":{"observed_at":"2026-08-07T14:50:00.957373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:04.231204Z","title":"ProsAudit, a prosodic benchmark for self-supervised speech models,","venue":null,"work_id":"e93a3e09-aa5f-4546-bb3b-cb264598cd2a","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.110052Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:8e0256cc6bbf5309ef64f97acb6aaa9ae5762cfc36257eee121c6b237abcdd7a","observation_id":"7877ab0c-d921-437a-85b6-25eff4f21cbc","resolution":{"observed_at":"2026-08-07T14:50:04.320994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.993226Z","title":"A corpus and cloze evaluation for deeper understanding of commonsense stories,","venue":null,"work_id":"48528b60-61ee-440b-858e-8e57a7cb7850","year":2016},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.237648Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ca78944fbd3580d7ed289ccc875eea172eb871f505907bef272ffc8912855589","observation_id":"688f3b9f-cfa1-49ac-9e5f-3dba2828f275","resolution":{"observed_at":"2026-08-07T14:50:04.110815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.755553Z","title":"Praat: doing phonetics by com- puter [computer program]. version 6.4.27,","venue":null,"work_id":"49d8caaf-285c-47aa-8f43-13ed58f780c2","year":2025},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.363667Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:6bd8a6b3938af663293e8e63e4c2f2e3de6f5fb1381a22f94e302fb528e446f5","observation_id":"5021f133-9058-4f65-ae6b-c35cee78d217","resolution":{"observed_at":"2026-08-07T14:50:03.857528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.482625Z","title":"Martinet, Elements of General Linguistics, ser","venue":null,"work_id":"c339d204-517d-4cc6-8d90-82ec1402f1c8","year":1966},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.513967Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:8d29ddf753a2a2b19eedb81402d010f045c755510ea9f84f8c71d02c396353f5","observation_id":"fc7b4e91-a9ec-48aa-a83e-67d0b3690784","resolution":{"observed_at":"2026-08-07T14:50:03.592334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.279265Z","title":"Are Discrete Units Nec- essary for Spoken Language Modeling?","venue":null,"work_id":"16bb16d1-3a0e-4dc5-a7f0-33fedd370949","year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.653076Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:991f91db56091f7a463d904edc2616f90f3354260499bfeaea294d2dc910799d","observation_id":"3805a6d2-15f9-4b42-9db8-6599ed20e1da","resolution":{"observed_at":"2026-08-07T14:50:03.373953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05755","last_updated":"2024-10-18T19:18:41Z","snapshot_observed_at":"2026-08-08T16:34:49.867190Z","submitted_at":"2024-02-08T15:39:32Z","title":"Spirit LM: Interleaved Spoken and Written Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05755","snapshot_observed_at":"2026-08-07T14:50:01.779226Z","title":"Spirit- lm: Interleaved spoken and written language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.779226Z"},"links":{"cited_paper":"/paper/2402.05755","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ea98b07688ca833a34222ae48e29829ffa9ae399a1014462f9d69dae1221002f","observation_id":"05e41d96-2af2-4730-a5b6-ac84e179efd6","resolution":{"observed_at":"2026-08-07T14:50:01.779226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.020545Z","title":"Multi- resolution hubert: Multi-resolution speech self-supervised learn- ing with masked unit prediction,","venue":null,"work_id":"560ec706-bf39-406d-b687-d7ba4849ff6c","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.870557Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ad3d3e439c1f93ea858b5cd47e37fd3c8bf9bf0f562a32a33aba2638b2e228eb","observation_id":"47015db1-6fa3-4ab6-99af-e34b107b2ff2","resolution":{"observed_at":"2026-08-07T14:50:03.120389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:02.749152Z","title":"Self-supervised contrastive learning for unsupervised phoneme segmentation,","venue":null,"work_id":"ee3424f7-a887-4b69-88d2-cb117ed57e08","year":2020},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.956551Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:a3cd62f8510018d3cfe4819aba0fe1681198071e8e27626386c0c0d39bada186","observation_id":"a782d3a6-28f9-42e3-ba2e-e69c11c1d4c7","resolution":{"observed_at":"2026-08-07T14:50:02.865867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:02.541294Z","title":"Unsupervised word segmentation using temporal gradient pseudo-labels,","venue":null,"work_id":"9162935d-dad0-4aae-8ad0-731b455b77cc","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:02.022181Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:29e82826d36df21704d6fdb833d4060b7cbb94caa71fdcb62df1fa99d94c48f2","observation_id":"abf54504-99ea-4786-be68-a93e630cec9a","resolution":{"observed_at":"2026-08-07T14:50:02.620676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T14:44:45.933263Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":1,"verified_fuzzy":30},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 1 inbound Pith citation observation for arXiv:2505.17446."}