{"as_of":"2026-08-23T17:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6c75063d1a81731e12969b7b766b12a2762c6c706e5a813c1132e34107091836","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T10:28:18.202974Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.28063/citation-record","integrity":"/paper/2605.28063/integrity","json":"/paper/2605.28063/citation-record.json","paper":"/paper/2605.28063"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-16T06:25:22.037199Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":"2412.10117","doi":"10.48550/arxiv.2412.10117","metadata_source":"pith","pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","venue":"cs.SD","work_id":"3af84775-3b81-4078-b553-52739aae03ba","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:578ba1ff44fb18faa5c4067c017d4985aae0e582ba37d462ad0932c53ae167ee","observation_id":"ece5bcda-246a-4892-8c05-8ffd6a467d9c","resolution":{"observed_at":"2026-06-29T13:23:28.661916Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-17T16:38:13.622895+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-17T16:38:13.622895+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-16T06:25:22.037199Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":"2412.10117","doi":"10.48550/arxiv.2412.10117","metadata_source":"pith","pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","venue":"cs.SD","work_id":"3af84775-3b81-4078-b553-52739aae03ba","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:4ffdd7c5805f5ac241a9a2febac1b3592cffdc84fdd84dbc64dafc3cd9a843b2","observation_id":"0d413b2f-096d-407f-9f01-caefb6e09dae","resolution":{"observed_at":"2026-06-29T10:33:17.883486Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-17T16:38:13.622895+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-17T16:38:13.622895+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17589","last_updated":"2025-05-27T07:48:34Z","snapshot_observed_at":"2026-08-16T06:12:24.457686Z","submitted_at":"2025-05-23T07:55:21Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","version":2},"cited_work":{"arxiv_id":"2505.17589","doi":"10.48550/arxiv.2505.17589","metadata_source":"pith","pith_arxiv_id":"2505.17589","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","venue":"cs.SD","work_id":"ce66e767-526a-419d-bb95-ac019edf4050","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2505.17589","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:ee7c6f0310caba4f0f3cbae3aa7409bea4f4334502095b25185e846efaaf0220","observation_id":"371d6347-b1b1-4839-9dc8-6ea913c27278","resolution":{"observed_at":"2026-06-29T10:33:17.880668Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-17T16:38:11.49925+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-17T16:38:11.49925+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-13T01:33:42.158497Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"cited_work":{"arxiv_id":"2601.15621","doi":"10.48550/arxiv.2601.15621","metadata_source":"pith","pith_arxiv_id":"2601.15621","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-TTS Technical Report","venue":"cs.SD","work_id":"6e1ae3e6-2062-4e44-a140-5c75409032cd","year":2026},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2601.15621","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:40b624fb3908dc30534ab827a82d0c674a4e505a319d2239ce78a8dc49869555","observation_id":"f534a123-42ed-408f-8bca-eecc927ea264","resolution":{"observed_at":"2026-06-29T10:33:17.875268Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-17T16:38:12.278123+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-17T16:38:12.278123+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Audiogen: Textually guided audio generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:7e1bb18c548e5f3c2204f9b77eb4f230df4dfe61fe6404c76ea99eb234e08c08","observation_id":"bfea9cc9-0dc8-4d42-8dc2-c4457ece23b4","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2024.339960","doi":"10.1109/taslp.2024.3399607","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Plumbley","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","work_id":"c8d1faa0-2667-4326-9700-4f9d7d38477c","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:08766fe6a2e6831c6b3aa6a7213db8c08bc5043b7a5f4ced51090226992551af","observation_id":"e04477e7-b9e9-4068-a968-b89a18add4f8","resolution":{"observed_at":"2026-06-29T10:33:17.869872Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9660.2025","doi":"10.1109/icassp49660.2025","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Masked image pretraining on language assisted representation","venue":null,"work_id":"c9f474fb-eb82-47dc-b6e2-4fca1225b9d3","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:84a8d8958366aad9f58ed0ae43d9539cded5bc40aaab4cbfee791a821108d92c","observation_id":"46e73937-64d8-44ac-9a81-f22cf4d1f8ab","resolution":{"observed_at":"2026-06-29T10:33:17.876811Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"8485.2024","doi":"10.1109/icassp48485.2024","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URL http://dx.doi.org/10.1109/ICASSP48485.2024.10447579","venue":null,"work_id":"c4a0500e-cb13-456d-825a-73ad45e75f95","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:bee91e9c1fe667c21f796abb1e081a115569e03c9bc49cc7ad1ac4beb1420b24","observation_id":"a9413d29-44bb-4b6b-9150-39d4fdc32e0b","resolution":{"observed_at":"2026-06-29T10:33:17.878156Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-18T11:51:16.869672+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T11:51:16.869672+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9660.2025","doi":"10.1109/icassp49660.2025","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Masked image pretraining on language assisted representation","venue":null,"work_id":"c9f474fb-eb82-47dc-b6e2-4fca1225b9d3","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:5e1e550ce798b05527dbfacfc0ba932cceba8b04c2936e89ea11a4a8b54ed783","observation_id":"e4932df2-b7c3-4b7e-adab-71f687d3de51","resolution":{"observed_at":"2026-06-29T10:33:17.928783Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08878","last_updated":"2026-04-20T12:44:20Z","snapshot_observed_at":"2026-07-06T22:32:13.055128Z","submitted_at":"2025-10-10T00:19:41Z","title":"ControlAudio: Tackling Text-Guided, Timing-Indicated and Intelligible Audio Generation via Progressive Diffusion Modeling","version":3},"cited_work":{"arxiv_id":"2510.08878","doi":"10.48550/arxiv.2510.08878","metadata_source":"pith","pith_arxiv_id":"2510.08878","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"ControlAudio: Tackling Text-Guided, Timing-Indicated and Intelligible Audio Generation via Progressive Diffusion Modeling","venue":"cs.SD","work_id":"2a5d3490-3ce9-4410-9101-4a877e38761c","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2510.08878","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:dbf3a9dbb6767fb9b3b231d6734fe62a1f4e709d875ff599ba94a54f1f859f29","observation_id":"b0d47770-c9c5-4768-85e5-a4b28b290007","resolution":{"observed_at":"2026-06-29T10:33:17.901185Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15821","last_updated":"2023-12-25T22:24:49Z","snapshot_observed_at":"2026-08-17T22:52:14.678531Z","submitted_at":"2023-12-25T22:24:49Z","title":"Audiobox: Unified Audio Generation with Natural Language Prompts","version":1},"cited_work":{"arxiv_id":"2312.15821","doi":"10.48550/arxiv.2312.15821","metadata_source":"pith","pith_arxiv_id":"2312.15821","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Audiobox: Unified audio generation with natural language prompts","venue":"cs.SD","work_id":"f7d2982e-a068-43ac-85f8-a5d3fb631fa2","year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2312.15821","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:918dd4ff2cb09ace9c9902e301bba19fd08a75a19963a97b28b72316299b9e59","observation_id":"444d884a-4575-45da-9dac-ac58abdf689f","resolution":{"observed_at":"2026-06-29T10:33:17.930200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2019-2441","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Weiss, Ye Jia, Zhifeng Chen, and Yonghui Wu","venue":null,"work_id":"82062014-09e6-4a32-9daa-dc3712dea0c5","year":2019},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:6a10fec61feefd88c0f316f92ba338dc1bd0d2b10ef63749993ccc79c5fb458e","observation_id":"b30c9e72-229f-4392-99ef-d8ffa8b9726a","resolution":{"observed_at":"2026-06-29T10:33:17.909901Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-13T09:49:57.294935+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T09:49:57.294935+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n19-1011","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A udio C aps: Generating captions for audios in the wild","venue":null,"work_id":"1ac1fac9-c7ba-4b7f-950f-1ad9b6bfe9e9","year":2019},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:2f7ffd9cd2ebe9376c62d9f8739eb1826c5a1984b92530b6a89f231332a8f0c0","observation_id":"0741283b-44c7-485a-b48f-3f377bb160fa","resolution":{"observed_at":"2026-06-29T10:33:17.911656Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Wavcaps: A chatgpt-assisted weakly-labelled audio captioning dataset for audio-language multimodal research,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:7434ea4426f1d5911a9c03324a0f1a25fe4e97dc25284d6f4b93325000f9743b","observation_id":"dfbf643f-4677-4750-9cc7-a5636ba2e02e","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2024.341944","doi":"10.1109/taslp.2024.3419446","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WavCaps: A ChatGPT-Assisted Weakly-Labelled Audio Captioning Dataset for Audio- Language Multimodal Research","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","work_id":"567c4705-d7b1-4037-81f9-291cf6eab890","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:d258f43560a56d411146b3ff98faf8d2753373f808b021c8729add5f24858201","observation_id":"57ac3978-4a6f-4105-9708-88906345c40e","resolution":{"observed_at":"2026-06-29T10:33:17.931394Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-12T01:20:07.31099+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T01:20:07.31099+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9660.2025","doi":"10.1109/icassp49660.2025","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Masked image pretraining on language assisted representation","venue":null,"work_id":"c9f474fb-eb82-47dc-b6e2-4fca1225b9d3","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:5b94007539ce9ed802db7a1338c669dba38b965fbf187492009a7f37e1f5b75c","observation_id":"0fec8409-fc71-4ffe-8518-67e7d803d2bf","resolution":{"observed_at":"2026-06-29T10:33:17.946571Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"6027.375517","doi":"10.1145/3746027.3755170","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Freeaudio: Training-free timing planning for controllable long-form text-to-audio generation,","venue":null,"work_id":"0ef52553-c60a-482e-a24f-bc5d363ef681","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:dd9574b6f6c43fc0807957104fb66dc1a4c1dc1ca175c17a0def9db6fcd16fb6","observation_id":"c1bc5f90-f798-4570-9dff-5051fbf68911","resolution":{"observed_at":"2026-06-29T10:33:17.901912Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9660.2025","doi":"10.1109/icassp49660.2025","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Masked image pretraining on language assisted representation","venue":null,"work_id":"c9f474fb-eb82-47dc-b6e2-4fca1225b9d3","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:648a19f03c25205d05335c54007325225e4d146380e70ac4814fa8358a8f87ad","observation_id":"4a491638-ff52-4895-90a5-e3ffaea17fac","resolution":{"observed_at":"2026-06-29T10:33:17.907854Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4647.368168","doi":"10.1145/3664647.3681680","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"model-predicted CoT","venue":null,"work_id":"ce2d50fa-3471-4f95-afa1-93839af6cfdd","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:73279732be1074c43af7f5fc46394af363017fd95ecf48ff9f04187e4ea8cef2","observation_id":"c8d226f0-b228-4cda-8687-c8de081ba039","resolution":{"observed_at":"2026-06-29T10:33:17.937711Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.04656","doi":"10.48550/arxiv.2601.04656","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flexivoice: Enabling flexible style control in zero- shot tts with natural language instructions.arXiv preprint arXiv:2601.04656,","venue":"arXiv (Cornell University)","work_id":"ee6caf16-dd17-407c-a0ee-a3e2058acd60","year":2026},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:c5d8b4ff8f731496397cd1a39d1b4dee11f7129633178a94bbed715fa9bbf25c","observation_id":"15ef98f5-b922-4d06-a182-15a3368a65c9","resolution":{"observed_at":"2026-06-29T10:33:17.913196Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11326","last_updated":"2025-08-15T08:53:56Z","snapshot_observed_at":"2026-08-17T11:34:17.139894Z","submitted_at":"2025-08-15T08:53:56Z","title":"MoE-TTS: Enhancing Out-of-Domain Text Understanding for Description-based TTS via Mixture-of-Experts","version":1},"cited_work":{"arxiv_id":"2508.11326","doi":"10.48550/arxiv.2508.11326","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.11326","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Moe-tts: Enhancing out-of-domain text understanding for description-based TTS via mixture-of-experts,","venue":null,"work_id":"7069e87a-9607-4df6-b66e-fcf2c140fba0","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2508.11326","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:542f7cb7cc7ec75bbd02b55d0ec6b9808d49e9bd0c81c33adde1cd4a51da4b86","observation_id":"5c51c5e4-1523-44d3-a286-74cf8ac7a6d0","resolution":{"observed_at":"2026-06-29T10:33:17.871334Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00704","last_updated":"2024-12-10T03:17:56Z","snapshot_observed_at":"2026-08-16T14:56:06.586855Z","submitted_at":"2023-10-01T15:49:46Z","title":"UniAudio: An Audio Foundation Model Toward Universal Audio Generation","version":6},"cited_work":{"arxiv_id":"2310.00704","doi":"10.48550/arxiv.2310.00704","metadata_source":"pith","pith_arxiv_id":"2310.00704","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Uniaudio: An audio founda- tion model toward universal audio generation","venue":"cs.SD","work_id":"32877994-1a46-4754-bfc9-a1b6f56ead50","year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2310.00704","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:5e9374982e71b267b21e62fb4207570409a639dac46676f5c09a55f4407d73c3","observation_id":"e2101993-a06f-4051-818d-136ce8d283bc","resolution":{"observed_at":"2026-06-29T10:33:17.934369Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Fugatto 1: Foundational generative audio transformer opus 1,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:70fbe86cc3b03701fb6f71c5d7f8175da16cbdbb5821210abb0f6226a1727316","observation_id":"d90cdc08-f892-4acc-9532-5626d9dd7b9e","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Available: https://openreview.net/forum?id=B2Fqu7Y2cd","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:77c039c0c11ee56694037fe55803df5b64fa67dac1bd13e80b4b893d97298627","observation_id":"312a9d9a-08e2-4680-8c09-50e62f97e871","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Chain-of-thought prompting elicits reasoning in large language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:c856d777590e9b9cd8165ca2d95eb05c5635d18a5667e4a28cc7084157d7bbb1","observation_id":"032ef048-05c2-49d6-bc39-abfb96283acf","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Cot-vtm: Visual-to-music generation with chain-of-thought reasoning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:90247e7a7074b14f83e4c94dcc30c13a0f02eec9ccf475d693f4e469c77d7f93","observation_id":"a7d00847-865e-4617-89f8-b018de417282","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"6027.375531","doi":"10.1145/3746027.3755316","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"In: Pro- ceedings of the 33rd ACM International Conference on Multimedia (ACM MM)","venue":null,"work_id":"e942c881-4e34-4f4d-8b17-5b13e912af46","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:11c8cdcdfa419d96f5a9d443751735b82919248f2685991fbf60e42ba3b5ad9b","observation_id":"a775bd8c-8635-4455-abaf-a63e67bd9e69","resolution":{"observed_at":"2026-06-29T10:33:17.922875Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.01459","doi":"10.48550/arxiv.2601.01459","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ov-instructtts: Towards open-vocabulary instruct text-to-speech.arXiv preprint arXiv:2601.01459,","venue":"arXiv (Cornell University)","work_id":"f1ba3c29-0039-410b-8b7d-8d31da7eda62","year":2026},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:2b1e3084b7c4cd79102c0fbaab9b204bacf182f570f54890e32674d85e6adf0b","observation_id":"4f459bcf-c373-403e-8527-2132986c57b2","resolution":{"observed_at":"2026-06-29T10:33:17.895673Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05171","last_updated":"2025-02-17T17:14:04Z","snapshot_observed_at":"2026-08-15T00:53:48.099058Z","submitted_at":"2025-02-07T18:55:02Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","version":2},"cited_work":{"arxiv_id":"2502.05171","doi":"10.1016/0041-5553(81)90075-6","metadata_source":"pith","pith_arxiv_id":"2502.05171","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","venue":"cs.LG","work_id":"1ee7474f-a930-486e-897c-207b8755f2c9","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2502.05171","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:00fcbaf33c381ae8872d87f5bcd6684fd24c9b9e7461375aad634fb7b3703daf","observation_id":"4f9adc71-74b0-4299-bd3e-49e20a41e968","resolution":{"observed_at":"2026-06-29T10:33:17.927585Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06769","last_updated":"2025-11-03T00:53:34Z","snapshot_observed_at":"2026-08-17T06:12:37.525438Z","submitted_at":"2024-12-09T18:55:56Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","version":3},"cited_work":{"arxiv_id":"2412.06769","doi":"10.48550/arxiv.2412.06769","metadata_source":"pith","pith_arxiv_id":"2412.06769","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","venue":"cs.CL","work_id":"3ddd0fd2-c176-408f-9b58-0666c2707f2d","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2412.06769","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:1fa1f0efa94101e3f55f6011bea1f3b2696d656f1e22bb9bc2898febaf32b0f1","observation_id":"69930641-b79b-4333-8aa6-b23ca2d6a1b4","resolution":{"observed_at":"2026-06-29T10:33:17.940347Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.16782","doi":"10.48550/arxiv.2505.16782","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reasoning beyond language: A comprehensive survey on latent chain-of-thought reasoning.arXiv preprint arXiv:2505.16782, 2025a","venue":"ArXiv.org","work_id":"3f6875b5-ba98-4d07-aaa0-c1ac8726244b","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:16a1c02878c745e5a4ecc5d1ef46b93b46ae825f48005435aeae188216d9c63a","observation_id":"53010181-7d25-4ccd-b174-76a5bbb84785","resolution":{"observed_at":"2026-06-29T10:33:17.922556Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-01T13:38:11.467762+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T13:38:11.467762+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08128","last_updated":"2025-07-28T22:53:43Z","snapshot_observed_at":"2026-08-17T22:20:20.496099Z","submitted_at":"2025-07-10T19:40:21Z","title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","version":2},"cited_work":{"arxiv_id":"2507.08128","doi":"10.48550/arxiv.2507.08128","metadata_source":"pith","pith_arxiv_id":"2507.08128","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","venue":"cs.SD","work_id":"67c2892d-8e27-4da1-8198-71c48e673e96","year":2025},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2507.08128","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:bebdd94c4a441bade56c8df0dda92742aff8c8a9c4c2b9b77df7ddb383bbc26e","observation_id":"a369a4ab-51fb-4b34-9c01-f58b2a0c16d6","resolution":{"observed_at":"2026-06-29T10:33:17.925822Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2017.795226","doi":"10.1109/icassp.2017.7952261","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemmeke, Daniel P","venue":null,"work_id":"5f4fb44f-6e07-4cb7-816f-efe601e93294","year":2017},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:b2d6f0887535557c2b370a16a6ba641cd51c33e78ca7c5f3267f84bead046ea3","observation_id":"6bb8d847-8a32-48fa-b849-63909e5dacac","resolution":{"observed_at":"2026-06-29T10:33:17.909848Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-13T14:20:41.234855+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T14:20:41.234855+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:40b8a32639f431c2271e8ab28f6639248d35baa9fe2c5922207d293f6e6d95dc","observation_id":"f1828948-1ee8-4a80-b90a-2366ec50d8d0","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Simple and controllable music generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:edd57976a0c224b11bcac7e1bc7233415954673b0de937ffbca5344027f60f65","observation_id":"cf62ae6b-102a-437f-8fa0-7184b8bbf937","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.13731","last_updated":"2023-05-29T12:09:08Z","snapshot_observed_at":"2026-08-16T15:38:16.919337Z","submitted_at":"2023-04-24T07:45:28Z","title":"Text-to-Audio Generation using Instruction-Tuned LLM and Latent Diffusion Model","version":2},"cited_work":{"arxiv_id":"2304.13731","doi":"10.48550/arxiv.2304.13731","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.13731","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Text-to-audio generation using instruction- tuned LLM and latent diffusion model","venue":"arXiv (Cornell University)","work_id":"492d96c3-fca9-41b7-a64d-3e3313ff9da1","year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"cited_paper":"/paper/2304.13731","citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:2cb824c9c38ee197b48365c2eb56730814dd2abbe2ff38b418cf350531c69dbb","observation_id":"08905fd2-4446-4e2b-ac4a-c5d44b33ca3a","resolution":{"observed_at":"2026-06-29T10:33:17.943293Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T10:28:18.202974Z","title":"Make-an-audio: Text-to-audio generation with prompt-enhanced diffusion models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:d855d89ec1d7333b89e81d582e476f45926b81937a29cc09a62b1851b7b4c67d","observation_id":"ec4ee320-d471-4e3b-aafb-48d439831d16","resolution":{"observed_at":"2026-06-29T10:28:18.202974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"8485.2024","doi":"10.1109/icassp48485.2024","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"URL http://dx.doi.org/10.1109/ICASSP48485.2024.10447579","venue":null,"work_id":"c4a0500e-cb13-456d-825a-73ad45e75f95","year":2024},"citing_paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T10:28:18.202974Z"},"links":{"citing_paper":"/paper/2605.28063"},"observation_digest":"sha256:010771a9739f3781602e5088ffab3f7577c24eb08a5470f0b67c664cd5ff4e78","observation_id":"77870d47-ec8f-40a9-8618-a66372ee6102","resolution":{"observed_at":"2026-06-29T10:33:17.925141Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-18T11:51:16.869672+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T11:51:16.869672+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.28063","last_updated":"2026-05-27T07:15:35Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-15T15:58:57.045936Z","submitted_at":"2026-05-27T07:15:35Z","title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":14,"parse_uncertain":0,"unresolved":9,"verified_exact":15,"verified_fuzzy":0},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 0 inbound Pith citation observations for arXiv:2605.28063."}