{"as_of":"2026-08-07T12:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:988a8984db625de165efbb11c07cf9e04b1326a6b60e2b5d49a847c5105a6740","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":50,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:45:29.628714Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T12:45:29.628714Z","title":"Xls-r: Self- supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.23619","last_updated":"2025-05-29T16:26:32Z","snapshot_observed_at":"2026-08-07T12:39:39.469021Z","submitted_at":"2025-05-29T16:26:32Z","title":"Few-Shot Speech Deepfake Detection Adaptation with Gaussian Processes","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:29.628714Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2505.23619"},"observation_digest":"sha256:0bd5c17a2f484111921e9ed80ba91dc326c5a8d1fb4711a665908591f21b76ae","observation_id":"8c6ee2ef-5abc-41d7-a341-6ff54fb01f02","resolution":{"observed_at":"2026-08-07T12:45:29.628714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T12:05:51.843072Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.00743","last_updated":"2025-05-31T23:09:26Z","snapshot_observed_at":"2026-08-07T11:56:43.158163Z","submitted_at":"2025-05-31T23:09:26Z","title":"Assortment of Attention Heads: Accelerating Federated PEFT with Head Pruning and Strategic Client Selection","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:05:51.843072Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.00743"},"observation_digest":"sha256:05cfde47ee0182cbb53169d955eb5e290e81e55cd164ef17fe32aa186c575937","observation_id":"ad37c472-53a9-44e6-baed-4e1850d22a49","resolution":{"observed_at":"2026-08-07T12:05:51.843072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T11:57:52.905967Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01014","last_updated":"2025-06-01T13:53:28Z","snapshot_observed_at":"2026-08-07T11:50:31.744649Z","submitted_at":"2025-06-01T13:53:28Z","title":"Rhythm Controllable and Efficient Zero-Shot Voice Conversion via Shortcut Flow Matching","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T11:57:52.905967Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.01014"},"observation_digest":"sha256:d7c85a74fb5a0546330a797d66c9d5ffa66e78a2ded15af40053fc804103d4d8","observation_id":"38d25a94-b079-4b34-b82c-8025f1e96b67","resolution":{"observed_at":"2026-08-07T11:57:52.905967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T05:43:08.358475Z","title":"Xls-r: Self- supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.07294","last_updated":"2025-08-16T22:33:43Z","snapshot_observed_at":"2026-08-07T05:34:41.391218Z","submitted_at":"2025-06-08T21:36:10Z","title":"Towards Generalized Source Tracing for Codec-Based Deepfake Speech","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:08.358475Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.07294"},"observation_digest":"sha256:d625a3eb49df4ffcf75740f9520813d813d1e9a2d8b7aa74516c7488da761b58","observation_id":"df060ada-c158-40dd-a58e-c297ca8c74c8","resolution":{"observed_at":"2026-08-07T05:43:08.358475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T04:32:30.227311Z","title":"Xls-r: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10349","last_updated":"2025-06-12T05:04:53Z","snapshot_observed_at":"2026-08-07T04:26:18.028137Z","submitted_at":"2025-06-12T05:04:53Z","title":"Joint ASR and Speaker Role Tagging with Serialized Output Training","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T04:32:30.227311Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.10349"},"observation_digest":"sha256:eeb5c8b40635e5e26e7e0ed96d74e5e8e2ec8c2f6fdce46abd65d112be047a6a","observation_id":"97d95895-7355-4b31-83b6-d50e6d76d3c4","resolution":{"observed_at":"2026-08-07T04:32:30.227311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T04:08:06.077547Z","title":"XLS-R: Self-supervised cross-lingual speech rep- resentation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.11532","last_updated":"2025-06-13T07:36:31Z","snapshot_observed_at":"2026-08-07T12:13:04.092318Z","submitted_at":"2025-06-13T07:36:31Z","title":"From Sharpness to Better Generalization for Speech Deepfake Detection","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:08:06.077547Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.11532"},"observation_digest":"sha256:c59bd376882efcb82f4b01e01dc16067968ee5147903875a83e624015b11b98e","observation_id":"e892ec6f-7acc-49c8-a6fa-fde74123eb15","resolution":{"observed_at":"2026-08-07T04:08:06.077547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-07T00:22:26.715238Z","title":"Xls-r: Self-supervised cross- lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.14153","last_updated":"2025-06-17T03:30:58Z","snapshot_observed_at":"2026-08-07T05:00:24.532348Z","submitted_at":"2025-06-17T03:30:58Z","title":"Pushing the Performance of Synthetic Speech Detection with Kolmogorov-Arnold Networks and Self-Supervised Learning Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:22:26.715238Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.14153"},"observation_digest":"sha256:7924a41cd42a1ab405b9a97048bc093549255ea7e4a63f2803cd282da4bbf5fc","observation_id":"ff93de16-16f8-419b-9e1a-54ae05522f3d","resolution":{"observed_at":"2026-08-07T00:22:26.715238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T23:35:28.171349Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.17459","last_updated":"2025-06-20T19:59:49Z","snapshot_observed_at":"2026-08-06T23:28:24.764934Z","submitted_at":"2025-06-20T19:59:49Z","title":"Breaking the Transcription Bottleneck: Fine-tuning ASR Models for Extremely Low-Resource Fieldwork Languages","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T23:35:28.171349Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.17459"},"observation_digest":"sha256:8cbd71d17590882099574c57a94707763bfaec2bede189c226c76ef6caf8198b","observation_id":"4ba1af66-387e-4ddc-bbca-aa4840297d0d","resolution":{"observed_at":"2026-08-06T23:35:28.171349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T23:34:49.009463Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.17525","last_updated":"2025-06-27T18:38:01Z","snapshot_observed_at":"2026-08-07T01:18:32.550933Z","submitted_at":"2025-06-21T00:34:18Z","title":"Data Quality Issues in Multilingual Speech Datasets: The Need for Sociolinguistic Awareness and Proactive Language Planning","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T23:34:49.009463Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2506.17525"},"observation_digest":"sha256:26d0f8d6f2a59ef52fed4f26067f5a3db70b62257d507bbb8a0882f0204103e4","observation_id":"bacce3da-fc5d-4cf5-bbbf-bfeef09b4d93","resolution":{"observed_at":"2026-08-06T23:34:49.009463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T21:07:11.272341Z","title":"Xls-r: Self-supervisedcross-lingual speech representation learning at scale","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.00951","last_updated":"2025-07-12T02:50:17Z","snapshot_observed_at":"2026-08-07T06:42:43.919500Z","submitted_at":"2025-07-01T16:52:25Z","title":"Thinking Beyond Tokens: From Brain-Inspired Intelligence to Cognitive Foundations for Artificial General Intelligence and its Societal Impact","version":3},"reference_index":159,"source":"pdf_text","source_observed_at":"2026-08-06T21:07:11.272341Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2507.00951"},"observation_digest":"sha256:004164716b8c7e96422dc321fd3d37d074d3fbc4dd81cdf8eefd01dc2caed579","observation_id":"045ac136-f4e3-4f8b-8dce-fd38e6707afd","resolution":{"observed_at":"2026-08-06T21:07:11.272341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T23:02:14.927874Z","title":"Xls-r: Self- supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02911","last_updated":"2025-06-25T00:39:33Z","snapshot_observed_at":"2026-08-06T22:55:03.121335Z","submitted_at":"2025-06-25T00:39:33Z","title":"DiceHuBERT: Distilling HuBERT with a Self-Supervised Learning Objective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:02:14.927874Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2507.02911"},"observation_digest":"sha256:94d40be70bc463c1708bcb7dcd32a41f647081926e8cb4892b8927655acea96b","observation_id":"dad648f9-9aa1-4934-b17b-0668ee6d121e","resolution":{"observed_at":"2026-08-06T23:02:14.927874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T19:44:26.885657Z","title":"Xls-r: Self- supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.04738","last_updated":"2025-07-07T08:10:26Z","snapshot_observed_at":"2026-08-06T19:38:18.857847Z","submitted_at":"2025-07-07T08:10:26Z","title":"Word stress in self-supervised speech models: A cross-linguistic comparison","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:44:26.885657Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2507.04738"},"observation_digest":"sha256:1bbb92cd115ea4cd6835f0ba912b4e214de6d90ee1b2ebe5dac4ee3216fdd9ec","observation_id":"845ae894-6d3c-49bf-8e4e-53e6d3f9e739","resolution":{"observed_at":"2026-08-06T19:44:26.885657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T20:01:35.492150Z","title":"Xls-r: Self- supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08012","last_updated":"2025-07-05T10:59:00Z","snapshot_observed_at":"2026-08-06T19:55:13.707476Z","submitted_at":"2025-07-05T10:59:00Z","title":"RepeaTTS: Towards Feature Discovery through Repeated Fine-Tuning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:01:35.492150Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2507.08012"},"observation_digest":"sha256:8d9c190b3e8c2243342394d9c79efbdc417b5e751212cf823b6e89ff1583cd88","observation_id":"6910d54b-696a-4824-9f91-cfc860254baf","resolution":{"observed_at":"2026-08-06T20:01:35.492150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T18:14:08.779769Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08768","last_updated":"2025-07-11T17:27:11Z","snapshot_observed_at":"2026-08-06T18:06:58.265245Z","submitted_at":"2025-07-11T17:27:11Z","title":"On Barriers to Archival Audio Processing","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T18:14:08.779769Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2507.08768"},"observation_digest":"sha256:adbc2365af7a12a67210211c218bcdbbe271398d2aa2c86831dd7a0e61c9a64f","observation_id":"1991a958-5633-48f9-ac24-634cc4e0fab1","resolution":{"observed_at":"2026-08-06T18:14:08.779769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-06T12:49:16.380954Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21463","last_updated":"2025-07-29T03:06:31Z","snapshot_observed_at":"2026-08-06T12:49:15.030804Z","submitted_at":"2025-07-29T03:06:31Z","title":"SpeechFake: A Large-Scale Multilingual Speech Deepfake Dataset Incorporating Cutting-Edge Generation Methods","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T12:49:16.380954Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2507.21463"},"observation_digest":"sha256:89e4bb68dcc3386f7b704b6cb0cf4e8ef0bab03e74f37d898ab7986b01d47e6f","observation_id":"f1c21fd1-9842-447b-8a1c-14db27d29faa","resolution":{"observed_at":"2026-08-06T12:49:16.380954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T15:39:21.516370Z","title":"Xls-r: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.19721","last_updated":"2025-08-27T09:30:43Z","snapshot_observed_at":"2026-08-06T18:41:16.320515Z","submitted_at":"2025-08-27T09:30:43Z","title":"CAM\\~OES: A Comprehensive Automatic Speech Recognition Benchmark for European Portuguese","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T15:39:21.516370Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2508.19721"},"observation_digest":"sha256:144e3e0ad03413fca3c938e274340eb958752ca9c6ad66a6a79dc3962b2b1775","observation_id":"ddfdcf68-6bbb-418c-92ca-b7e5ef60aa85","resolution":{"observed_at":"2026-08-05T15:39:21.516370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T13:55:18.277810Z","title":"Xls-r: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T05:00:32.890157Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.277810Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:b45bbdea75879385787750bf9de1dcc8e1fddbb280c1fe41351c5e19cdeb2c95","observation_id":"b1e5f998-c1c8-4785-91e5-dd8b66b957c6","resolution":{"observed_at":"2026-08-05T13:55:18.277810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2510.02864","last_updated":"2026-05-06T07:16:22Z","snapshot_observed_at":"2026-07-06T22:31:35.666707Z","submitted_at":"2025-10-03T10:02:34Z","title":"Forensic Similarity for Speech Deepfakes","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T10:28:12.467606Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2510.02864"},"observation_digest":"sha256:bd05ac7b1499614c26ef0c895f56b9e151c58df5a2e95f18d4e1e4bf533829a0","observation_id":"b423e264-8805-408e-8884-c868e8f5ad10","resolution":{"observed_at":"2026-05-18T10:31:14.902664Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-03T20:05:29.992627Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2511.21325","last_updated":"2026-07-19T05:59:03Z","snapshot_observed_at":"2026-08-07T11:07:47.709808Z","submitted_at":"2025-11-26T12:16:38Z","title":"SONAR: Spectral-Contrastive Audio Residuals for Generalizable Deepfake Detection","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-03T20:05:29.992627Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2511.21325"},"observation_digest":"sha256:03bc4268262b3848a5eef0dcd0653aeed5ed70dae5ec6e9f7c9e81a012ba238a","observation_id":"7ea9d4ba-0fda-47c3-ada1-5f2e3c64aae3","resolution":{"observed_at":"2026-08-03T20:05:29.992627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2603.01482","last_updated":"2026-03-02T05:45:55Z","snapshot_observed_at":"2026-07-06T22:47:28.681790Z","submitted_at":"2026-03-02T05:45:55Z","title":"A SUPERB-Style Benchmark of Self-Supervised Speech Models for Audio Deepfake Detection","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T17:25:27.137423Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2603.01482"},"observation_digest":"sha256:169413f9db086d16ae27de86d5c1c74babe51271e31af080f61a9b40f932be18","observation_id":"98b8ee19-d213-48f5-9aa8-f181363ba2b3","resolution":{"observed_at":"2026-05-15T17:26:22.179791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2604.04598","last_updated":"2026-04-06T11:23:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T11:23:42Z","title":"Benchmarking Multilingual Speech Models on Pashto: Zero-Shot ASR, Script Failure, and Cross-Domain Evaluation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T19:44:30.762851Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2604.04598"},"observation_digest":"sha256:fbc82a38fdc437785076a6602784dc0927a2e8499f0d8f3a622e19f5c0e3e4cb","observation_id":"be6fd803-7145-4683-a19c-101dd423dd21","resolution":{"observed_at":"2026-05-10T22:35:48.700833Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2604.13288","last_updated":"2026-04-14T20:32:29Z","snapshot_observed_at":"2026-08-02T20:31:56.427045Z","submitted_at":"2026-04-14T20:32:29Z","title":"Giving Voice to the Constitution: Low-Resource Text-to-Speech for Quechua and Spanish Using a Bilingual Legal Corpus","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T15:01:20.902986Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2604.13288"},"observation_digest":"sha256:99ea0dfcf63ed09e4874272d3b70e104aaac7465ef611adc7f8b7299c403869b","observation_id":"db07feee-6407-4425-8621-09c452858f2c","resolution":{"observed_at":"2026-05-11T11:21:00.993257Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2604.26057","last_updated":"2026-04-28T18:52:38Z","snapshot_observed_at":"2026-07-06T23:11:48.524898Z","submitted_at":"2026-04-28T18:52:38Z","title":"Similarity Choice and Negative Scaling in Supervised Contrastive Learning for Deepfake Audio Detection","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-07T13:50:53.553261Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2604.26057"},"observation_digest":"sha256:7afdbfb048d144ba9b5ab036d4ded54d67d88c12e21e605525a953c2f9d3e51e","observation_id":"dbb3df5e-639c-4e03-a9f6-4c0a3e50176d","resolution":{"observed_at":"2026-05-12T08:46:25.862041Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2605.17737","last_updated":"2026-05-18T01:36:46Z","snapshot_observed_at":"2026-08-02T07:35:19.943282Z","submitted_at":"2026-05-18T01:36:46Z","title":"Profiling the Voice: Speaker-Specific Phoneme Fingerprinting for Speech Deepfake Detection","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T01:20:04.609575Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2605.17737"},"observation_digest":"sha256:2d345c627199fe61066effacd369daecaab3a8698de8edcbdc34d6ddb84f0640","observation_id":"e98638b2-ad8c-42d9-8f32-79e8ac4438e6","resolution":{"observed_at":"2026-05-20T01:22:55.817247Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2605.23201","last_updated":"2026-05-22T03:33:36Z","snapshot_observed_at":"2026-07-06T23:33:24.215365Z","submitted_at":"2026-05-22T03:33:36Z","title":"MixFake: Benchmarking and Enhancing Audio Deepfake Detection in Diverse Real-world Mixed Audio","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-25T03:14:08.880072Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2605.23201"},"observation_digest":"sha256:37f4542ed7eccfb87dd91c87e4831291b8041d8dabb8f3396dca32dff8914d54","observation_id":"cce6bf73-5148-4a1d-b5a5-29cb6cbcb3d4","resolution":{"observed_at":"2026-05-25T03:15:17.023264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2605.30366","last_updated":"2026-05-18T03:43:41Z","snapshot_observed_at":"2026-08-04T20:00:25.137382Z","submitted_at":"2026-05-18T03:43:41Z","title":"Escaping the Linearity Trap: Manifold Detours for Black-Box Adversarial Attacks on Singing Audio Deepfake Detection","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-30T18:58:31.318883Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2605.30366"},"observation_digest":"sha256:18a01b6280c13fa4110b836e9b47f673b16ba97dac6b220bca65a68742b528d9","observation_id":"ddaf9454-c96e-42c0-b5fe-c0ab99677897","resolution":{"observed_at":"2026-06-30T19:05:01.112227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.10223","last_updated":"2026-06-08T22:22:48Z","snapshot_observed_at":"2026-07-06T23:49:27.565372Z","submitted_at":"2026-06-08T22:22:48Z","title":"Dual-Branch Gated Fusion for Open-Set Audio Deepfake Source Tracing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T14:45:45.359616Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.10223"},"observation_digest":"sha256:c578262978e3948d69a4f18667b93e98992556c840265a2e18ba521c6c38b4d9","observation_id":"92c547c6-ca7c-4952-967d-fb9b26bfd232","resolution":{"observed_at":"2026-07-03T03:47:35.572305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.11542","last_updated":"2026-06-10T01:07:32Z","snapshot_observed_at":"2026-08-04T09:55:13.088817Z","submitted_at":"2026-06-10T01:07:32Z","title":"Pretrained self-supervised speech models can recognize unseen consonants","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T10:14:47.932613Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.11542"},"observation_digest":"sha256:9ce2148bf977aefa83486a0ca9352798a5eb64fbb5baf1988b46be3faa6a929f","observation_id":"f4260727-f010-4480-8756-d5f2a3dee562","resolution":{"observed_at":"2026-07-03T10:07:56.070600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.11674","last_updated":"2026-06-10T05:36:53Z","snapshot_observed_at":"2026-08-07T08:50:24.374833Z","submitted_at":"2026-06-10T05:36:53Z","title":"SpAArSIST: Sparsified AASIST for Efficient and Reliable Anti-Spoofing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T08:35:24.321197Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.11674"},"observation_digest":"sha256:9138866b625012a73d28cbddb525041c9b72368172689aaf3405ca73c12ba215","observation_id":"cbe71276-4f07-4fdb-8176-4a33c86cc761","resolution":{"observed_at":"2026-07-03T12:58:08.668281Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.18659","last_updated":"2026-06-17T04:00:06Z","snapshot_observed_at":"2026-07-30T02:31:06.367221Z","submitted_at":"2026-06-17T04:00:06Z","title":"Responsible ASR: Overcoming Challenges of Foundational Models in Narrow-Band and Low-Resource Settings","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T20:03:28.491546Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.18659"},"observation_digest":"sha256:7f05409cb748ac805ed06af1271e57f85874f67bbd816268bdfddcd915db14e0","observation_id":"18ed6e2d-fcd1-4e0b-9178-354240c9df74","resolution":{"observed_at":"2026-07-04T01:59:26.658181Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.19910","last_updated":"2026-06-23T10:40:32Z","snapshot_observed_at":"2026-08-05T15:52:37.159854Z","submitted_at":"2026-06-18T08:04:16Z","title":"Light-weight Pronunciation Assessment via Discrete Speech Token Surprisal","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-26T17:37:25.043607Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.19910"},"observation_digest":"sha256:0c5a9ddb63655256239be81d0bcf02b79ab46770cda185d75518de767b904464","observation_id":"dd75c079-7ec4-4819-9099-12cd136d5123","resolution":{"observed_at":"2026-07-04T03:49:30.231851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.20478","last_updated":"2026-06-18T16:55:44Z","snapshot_observed_at":"2026-08-06T22:35:47.074866Z","submitted_at":"2026-06-18T16:55:44Z","title":"Beyond Speaker Independence: Evaluating Cross-Lingual Acoustic-to-Articulatory Inversion Across Finnish and Russian","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T15:24:04.594993Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.20478"},"observation_digest":"sha256:d52c0075e03ecd88a4b81ee36a22051a976d9966587a91695d0e5d3f443d50bb","observation_id":"dedb92c1-9f15-4ce5-8432-63b950b9a2fe","resolution":{"observed_at":"2026-07-04T05:49:37.763673Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.21210","last_updated":"2026-06-26T13:32:46Z","snapshot_observed_at":"2026-07-06T23:56:17.293417Z","submitted_at":"2026-06-19T08:26:11Z","title":"Impact Analysis of Speech Representation Learning Models for Acoustic Side-Channel Attack","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-26T14:19:59.573391Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.21210"},"observation_digest":"sha256:28427fc9f0e04964fc54f815169325d3e6bdd5948fa4b0bddfc0f55e37240501","observation_id":"e82d9116-99fc-4ed9-b1a5-7b70cbc47ce5","resolution":{"observed_at":"2026-07-04T06:39:37.331175Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.21210","last_updated":"2026-06-26T13:32:46Z","snapshot_observed_at":"2026-07-06T23:56:17.293417Z","submitted_at":"2026-06-19T08:26:11Z","title":"Impact Analysis of Speech Representation Learning Models for Acoustic Side-Channel Attack","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T04:44:52.537405Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.21210"},"observation_digest":"sha256:a45a1536cb8c23391487a03ca0af2075505c897d31c968edb7edb1593659f428","observation_id":"fbff2c09-d503-4660-b16d-78a7dabb35c4","resolution":{"observed_at":"2026-06-29T19:33:54.406475Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.22274","last_updated":"2026-06-20T23:51:55Z","snapshot_observed_at":"2026-08-06T13:44:44.139123Z","submitted_at":"2026-06-20T23:51:55Z","title":"From Speech to Text Corpora: Evaluating ASR-Based Data Acquisition for Low-Resource Fongbe and Hausa","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T11:31:09.747113Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.22274"},"observation_digest":"sha256:48989d8d40504823074b31048ab5cc14f68f3cd95140d37d002df2009841a10f","observation_id":"61adb2a8-dc23-40af-a1f2-68848c557b19","resolution":{"observed_at":"2026-07-04T08:29:42.395446Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.26384","last_updated":"2026-06-24T21:10:42Z","snapshot_observed_at":"2026-08-02T18:27:05.790496Z","submitted_at":"2026-06-24T21:10:42Z","title":"What Do Deepfake Benchmarks Measure? An Audit Using Frozen Self-Supervised Representations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T01:24:46.200846Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.26384"},"observation_digest":"sha256:3b9aa0347f6ce83fc439823c99bb378cba3530b318a4de18a1fa19579bbbaf59","observation_id":"5d125a69-5efe-41d1-b65a-f8de9b2f395c","resolution":{"observed_at":"2026-07-04T15:49:57.200451Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.27206","last_updated":"2026-06-25T16:02:14Z","snapshot_observed_at":"2026-08-02T08:21:01.224967Z","submitted_at":"2026-06-25T16:02:14Z","title":"Syntactic Belief Update as the Driver of Garden Path Processing Difficulty","version":1},"reference_index":295,"source":"arxiv_source","source_observed_at":"2026-06-26T04:38:01.183423Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.27206"},"observation_digest":"sha256:34dce59135cc3039126cb9f0cfdc1958e147ca5e5fbe221d64c3eaeb92ddedd2","observation_id":"2a86893c-fa6b-413d-9c74-b2607ed4d8f7","resolution":{"observed_at":"2026-06-26T04:38:58.354104Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2606.30780","last_updated":"2026-06-29T18:09:27Z","snapshot_observed_at":"2026-08-02T20:33:10.214133Z","submitted_at":"2026-06-29T18:09:27Z","title":"Detecting Audio Deepfakes on the Edge:Lightweight SSL-Based Detection in a Browser Plugin","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-01T01:47:41.684335Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2606.30780"},"observation_digest":"sha256:be6989b943ebe3e462c001ecd01328e838e9af96ee064317a5fd220c88500cd6","observation_id":"5e46740a-119f-4c26-ac54-9e04ccec34c6","resolution":{"observed_at":"2026-07-01T12:45:44.602950Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":"2111.09296","doi":"10.48550/arxiv.2111.09296","metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XLS-R: Self-supervised cross-lingual speech represen- tation learning at scale","venue":"arXiv (Cornell University)","work_id":"c6922371-271e-4ae1-99f4-da2e47c9ed0b","year":2021},"citing_paper":{"arxiv_id":"2607.00387","last_updated":"2026-07-01T03:32:08Z","snapshot_observed_at":"2026-08-01T10:46:33.437238Z","submitted_at":"2026-07-01T03:32:08Z","title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-07-02T05:52:55.818877Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.00387"},"observation_digest":"sha256:e2b0cf8882f5bc697a67276eae4287d47a3a60edfc22424920172302032b44f6","observation_id":"abbf0d63-cf32-495b-a9cb-570350d25ac4","resolution":{"observed_at":"2026-07-02T05:56:39.846458Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"XLS-R: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-02T09:00:28.263806Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:3458fb796131fcafd2f65f571b0b9a126368c7482137adce13293da3252ee6ba","observation_id":"5a231095-8343-43cd-9b28-7b84916812e4","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-12T04:31:11.771594Z","title":"Xls-r: Self-supervised cross- lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.03150","last_updated":"2026-07-03T09:42:31Z","snapshot_observed_at":"2026-08-03T21:14:30.935788Z","submitted_at":"2026-07-03T09:42:31Z","title":"An Intervention-Based Framework for Shortcut Diagnosis in Spoofing Countermeasures","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-12T04:31:11.771594Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.03150"},"observation_digest":"sha256:1aaf78e173ad23318a6dfef13904fc6128bdab659ea60f5072908027a486053a","observation_id":"59a919b8-fb23-4828-adb2-3d3995836e7c","resolution":{"observed_at":"2026-07-12T04:31:11.771594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-11T18:11:36.626189Z","title":"Xls-r: Self-supervised cross- lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.04515","last_updated":"2026-07-05T21:37:15Z","snapshot_observed_at":"2026-08-07T10:46:12.581579Z","submitted_at":"2026-07-05T21:37:15Z","title":"Towards Digital Preservation of Efik: TTS for a Low-Resource African Language","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-11T18:11:36.626189Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.04515"},"observation_digest":"sha256:b4bbb51efa81940ae4c3255d4a6d7a1f698a106acbceb2f3e9a53d51333ccd5c","observation_id":"5265bd2a-3338-4e30-b7ee-b6840c656f53","resolution":{"observed_at":"2026-07-11T18:11:36.626189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-11T13:07:13.175134Z","title":"2111.09296 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04814","last_updated":"2026-07-27T07:41:53Z","snapshot_observed_at":"2026-08-02T08:34:53.822828Z","submitted_at":"2026-07-06T08:49:02Z","title":"Evaluating the Effect of Linguistic Relatedness on Cross-Lingual Transfer in Large Multilingual Automatic Speech Recognition","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-07-11T13:07:13.175134Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.04814"},"observation_digest":"sha256:81a05eafc65fd7829fb044578f127fa7898b0917227b97142343bbdcdcad659d","observation_id":"83eed9c1-9c22-4fb8-b782-3ff5b7d21eb3","resolution":{"observed_at":"2026-07-11T13:07:13.175134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-02T08:34:55.605107Z","title":"2111.09296 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04814","last_updated":"2026-07-27T07:41:53Z","snapshot_observed_at":"2026-08-02T08:34:53.822828Z","submitted_at":"2026-07-06T08:49:02Z","title":"Evaluating the Effect of Linguistic Relatedness on Cross-Lingual Transfer in Large Multilingual Automatic Speech Recognition","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T08:34:55.605107Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.04814"},"observation_digest":"sha256:ecee23613a10c9938623ebe5e152e2635698d6aa5edaf24be6b056493c652661","observation_id":"087998ef-6c92-4171-8258-2dacc3719058","resolution":{"observed_at":"2026-08-02T08:34:55.605107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-14T12:15:06.470439Z","title":"XLS-R: Self-supervised cross- lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.10371","last_updated":"2026-07-11T15:48:03Z","snapshot_observed_at":"2026-08-07T10:45:20.188896Z","submitted_at":"2026-07-11T15:48:03Z","title":"GigaAM Multilingual: Foundation Model for Underrepresented Languages","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-14T12:15:06.470439Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.10371"},"observation_digest":"sha256:0a3848ceb97dc5c1dce8188a765a945ce09179a23df1f8fa065586673d57326a","observation_id":"2d8c6bc4-790e-437b-acb7-3688e45872b3","resolution":{"observed_at":"2026-07-14T12:15:06.470439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-14T06:34:55.089754Z","title":"Xls-r: Self-supervised cross- lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.11163","last_updated":"2026-07-13T06:57:02Z","snapshot_observed_at":"2026-08-02T05:26:21.947350Z","submitted_at":"2026-07-13T06:57:02Z","title":"Unified Gradient Projection: Language-Balanced Continual Learning for Multilingual Low-Resource ASR","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-14T06:34:55.089754Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.11163"},"observation_digest":"sha256:a19ed681d48c7083fbeb0dd45ab69cdcd675902a52ee883448e16c7a5da30c15","observation_id":"0b197b28-1724-4dbe-9a52-d2703304a7d9","resolution":{"observed_at":"2026-07-14T06:34:55.089754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-01T07:10:39.837818Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.21540","last_updated":"2026-07-24T18:05:20Z","snapshot_observed_at":"2026-08-07T09:16:01.540251Z","submitted_at":"2026-07-23T17:25:08Z","title":"DONDO: Open w2v-BERT Speech-Recognition Base Models for African Languages","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T07:10:39.837818Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.21540"},"observation_digest":"sha256:58123987cbbd5b79f03934f4a9beb933a9fe4d49222fe3aab144d89f9c96754c","observation_id":"397ebc1a-c572-4365-b947-30a876530d83","resolution":{"observed_at":"2026-08-01T07:10:39.837818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-31T23:27:02.078077Z","title":"Xls-r: Self- supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.23961","last_updated":"2026-07-27T03:17:21Z","snapshot_observed_at":"2026-08-06T05:12:12.232358Z","submitted_at":"2026-07-27T03:17:21Z","title":"Leveraging Gradient Reversal Loss and Multitask Learning for Datasets-Aware Audio Deepfake Detection","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-31T23:27:02.078077Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.23961"},"observation_digest":"sha256:250acc0ef0f02e11c180d79dce00e0e080c0802f887d9fc14134b82b7b87b41f","observation_id":"af26c9ad-3b23-4deb-be16-701e9507960f","resolution":{"observed_at":"2026-07-31T23:27:02.078077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-31T10:28:22.087327Z","title":"Xls-r: Self-supervised cross-lingual speech representation learning at scale.arXiv preprint arXiv:2111.09296,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28351","last_updated":"2026-07-30T15:21:38Z","snapshot_observed_at":"2026-08-06T13:13:23.062215Z","submitted_at":"2026-07-30T15:21:38Z","title":"Teffic-Audio: Tell Fact from Fiction","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-31T10:28:22.087327Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.28351"},"observation_digest":"sha256:08033b10a45c2afbd2e98c0b56277ed83dc50f58f24c5a6c224c996325124372","observation_id":"46196a50-20d5-4b09-b76d-b2dd78425e91","resolution":{"observed_at":"2026-07-31T10:28:22.087327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-04T20:44:55.686214Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.01796","last_updated":"2026-08-03T07:04:51Z","snapshot_observed_at":"2026-08-06T23:33:06.192700Z","submitted_at":"2026-08-03T07:04:51Z","title":"Multi-Backbone Self-Supervised Ensembles for Audio Deepfake Detection and a Cross-Track Analysis of Generation-Detection Asymmetry","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T20:44:55.686214Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2608.01796"},"observation_digest":"sha256:9a077b223c27919274f1c13e76d0371622c2914e70fe9dbc4a288c73a8277b0f","observation_id":"d97e4775-c32e-4e8e-a2a1-0ac095920b20","resolution":{"observed_at":"2026-08-04T20:44:55.686214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2111.09296/citation-record","integrity":"/paper/2111.09296/integrity","json":"/paper/2111.09296/citation-record.json","paper":"/paper/2111.09296"},"outbound":[],"paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 50 inbound Pith citation observations for arXiv:2111.09296."}