{"as_of":"2026-08-05T08:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e94971f8ec6c1ce803d76954efcd3729edd8b4502d302f7ad475006edfdb0471","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":14,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":14,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":14,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":14,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T13:29:53.404566Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T02:49:24.877685Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:7c1fba04229d3c979f53f7a6813f964c4113e8905145e7068280c51749c111e1","observation_id":"5c555d22-4377-4e1e-b64f-0cc68308a95a","resolution":{"observed_at":"2026-05-16T05:59:51.111664Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2509.17765","last_updated":"2025-09-22T13:26:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-22T13:26:24Z","title":"Qwen3-Omni Technical Report","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-11T00:20:37.406351Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2509.17765"},"observation_digest":"sha256:9e8b08db51bd818d1333bc646a1ba45c1ead49391e4ce6e58bd276498fa2af22","observation_id":"f404586a-99f0-435a-8b3f-92b8deed627d","resolution":{"observed_at":"2026-05-11T00:20:38.178397Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-04T12:22:34.670840Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T19:24:56.057631Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2601.15621"},"observation_digest":"sha256:09975d06e89a32d0c5a49e294c7cea269453c8d77632d746f7ce8908dcf4eb6c","observation_id":"8bb46fa4-e9f3-4ddb-b5ca-0fde45dc48ab","resolution":{"observed_at":"2026-05-16T19:24:56.108210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2603.25551","last_updated":"2026-04-06T15:20:14Z","snapshot_observed_at":"2026-07-06T22:50:41.736941Z","submitted_at":"2026-03-26T15:23:34Z","title":"Voxtral TTS","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T00:38:42.441340Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2603.25551"},"observation_digest":"sha256:32c214d1e8b3127d47715b8a0f9200b94f8d15d96227abaa0a90254dc1b72a1e","observation_id":"0337724a-9c59-481c-9ea8-1557a43548e9","resolution":{"observed_at":"2026-05-15T00:39:35.876040Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2604.00688","last_updated":"2026-04-21T12:14:57Z","snapshot_observed_at":"2026-07-06T22:51:26.754746Z","submitted_at":"2026-04-01T09:45:51Z","title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T23:00:18.720371Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2604.00688"},"observation_digest":"sha256:b398bcdecbb7f22f96e91a0d5bd9c06bc494d5a55a3ecfd8b963f3504b858d4e","observation_id":"bf3233e1-8325-4ee6-bf24-e01632e14d75","resolution":{"observed_at":"2026-05-13T23:03:24.769453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2604.21164","last_updated":"2026-04-27T17:57:41Z","snapshot_observed_at":"2026-07-06T23:07:47.244208Z","submitted_at":"2026-04-23T00:13:16Z","title":"MAGIC-TTS: Fine-Grained Controllable Speech Synthesis with Explicit Local Duration and Pause Control","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T13:49:09.386368Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2604.21164"},"observation_digest":"sha256:ce7c2b2d3a6a80021f5dcca2d341c5212fe04622936fd03d6918ccbbd20a8507","observation_id":"a265c1c4-9ce2-46b3-87c6-f95636f518df","resolution":{"observed_at":"2026-05-11T18:51:06.774413Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2604.23586","last_updated":"2026-08-04T09:15:37Z","snapshot_observed_at":"2026-08-05T07:47:16.138769Z","submitted_at":"2026-04-26T07:48:47Z","title":"Talker-T2AV: Joint Talking Audio-Video Generation with Autoregressive Diffusion Modeling","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-08T06:44:36.000353Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2604.23586"},"observation_digest":"sha256:515bfd2f2bbb147254ee3a12c9495582f7281ea45f998f12fd215bac95cdaee3","observation_id":"bbcf6cd4-b43c-452a-bc91-d6fc1e178eba","resolution":{"observed_at":"2026-05-11T21:06:14.063829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2605.22083","last_updated":"2026-07-17T07:00:02Z","snapshot_observed_at":"2026-08-02T13:29:49.967308Z","submitted_at":"2026-05-21T07:22:28Z","title":"RobustSpeechFlow: Learning Robust Text-to-Speech Trajectories via Augmentation-based Contrastive Flow Matching","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-22T02:56:06.910758Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2605.22083"},"observation_digest":"sha256:c96022ca0cfe69e95d57a9b957bf256fe0aa4a792135f932e885f9573a9d2276","observation_id":"73bb91e9-1be7-423b-b39d-a6e3c88be29a","resolution":{"observed_at":"2026-05-22T03:00:58.977726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-08-02T13:29:53.404566Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.22083","last_updated":"2026-07-17T07:00:02Z","snapshot_observed_at":"2026-08-02T13:29:49.967308Z","submitted_at":"2026-05-21T07:22:28Z","title":"RobustSpeechFlow: Learning Robust Text-to-Speech Trajectories via Augmentation-based Contrastive Flow Matching","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T13:29:53.404566Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2605.22083"},"observation_digest":"sha256:4dc0998367c521f4985bf4abfd90e82fa05facc3ea08cecb39296d72831770d6","observation_id":"f036e8b3-a8aa-40dc-9b07-c05582b7026b","resolution":{"observed_at":"2026-08-02T13:29:53.404566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-13T08:22:44.347941Z","title":"cc/paper_files/paper/2019/file/1e8a19426224ca89e83cef47f1e7f53b-Paper.pdf","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2605.23912","last_updated":"2026-04-08T23:43:46Z","snapshot_observed_at":"2026-08-03T04:32:52.637811Z","submitted_at":"2026-04-08T23:43:46Z","title":"Raon-Speech Technical Report","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-13T08:22:44.347941Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2605.23912"},"observation_digest":"sha256:b737bef563caf6c0dbe2d3dcd7129b31c343eb114f87ea6e43a122bc6e2817c9","observation_id":"7a9da86e-b539-440f-b7d0-d589bed79e3e","resolution":{"observed_at":"2026-07-13T08:22:44.347941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2605.27258","last_updated":"2026-05-26T16:36:56Z","snapshot_observed_at":"2026-07-06T23:36:58.541051Z","submitted_at":"2026-05-26T16:36:56Z","title":"PilotTTS: A Disciplined Modular Recipe for Competitive Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T15:51:21.519785Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2605.27258"},"observation_digest":"sha256:c18c2faf390553a88f3630910701d9cc36a8c7214d745a9ebfc036010d1ddf1b","observation_id":"15cdf836-4114-4b60-96f9-fb7dbc455bfa","resolution":{"observed_at":"2026-06-29T16:23:39.874719Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2606.06928","last_updated":"2026-06-05T05:43:15Z","snapshot_observed_at":"2026-07-06T23:46:37.903443Z","submitted_at":"2026-06-05T05:43:15Z","title":"VoxCPM2 Technical Report","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T21:18:22.911332Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2606.06928"},"observation_digest":"sha256:03cb6758ed3a767aaa6d779d42c7598bc0ae514a306fd298f7e33134433aefb2","observation_id":"69e13e97-23ab-42e5-8719-a38d336cc674","resolution":{"observed_at":"2026-07-02T19:47:19.778528Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2606.19325","last_updated":"2026-06-17T17:51:50Z","snapshot_observed_at":"2026-08-03T14:31:26.558590Z","submitted_at":"2026-06-17T17:51:50Z","title":"Reference-Driven Multi-Speaker Audio Scene Generation from In-the-Wild Priors","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-26T19:04:25.542629Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2606.19325"},"observation_digest":"sha256:09f0f8595a267f87e46472604000b4d8a4a5ebde4f4320085c29cb6ac56e14c2","observation_id":"50e092b2-dc59-4516-9d83-940688b066fd","resolution":{"observed_at":"2026-07-04T02:49:24.879690Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","version":1},"cited_work":{"arxiv_id":"2505.07916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.07916","snapshot_observed_at":"2026-07-04T02:49:24.877685Z","title":"Minimax- speech: Intrinsic zero-shot text-to-speech with a learnable speaker encoder","venue":null,"work_id":"54273fa8-90e2-4bbd-8703-905eed26289f","year":2025},"citing_paper":{"arxiv_id":"2606.20650","last_updated":"2026-06-08T06:54:55Z","snapshot_observed_at":"2026-08-02T23:08:38.292740Z","submitted_at":"2026-06-08T06:54:55Z","title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T17:01:13.972071Z"},"links":{"cited_paper":"/paper/2505.07916","citing_paper":"/paper/2606.20650"},"observation_digest":"sha256:47439fc4743420262e85ece80271639709e3cceda8a07b2383db4eba8c8477bc","observation_id":"68de5465-d55c-4082-8340-0bfc92d14875","resolution":{"observed_at":"2026-07-03T00:47:30.528359Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.07916/citation-record","integrity":"/paper/2505.07916/integrity","json":"/paper/2505.07916/citation-record.json","paper":"/paper/2505.07916"},"outbound":[],"paper":{"arxiv_id":"2505.07916","last_updated":"2025-05-12T14:25:20Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-07-06T21:22:45.982350Z","submitted_at":"2025-05-12T14:25:20Z","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 14 inbound Pith citation observations for arXiv:2505.07916."}