{"as_of":"2026-08-11T15:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d21b3f9cc1d236f1a0e91c461a7eb90bdaf0dd8827fea244a1a48ecf57ec2eae","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:27:22.494754Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":25,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-10T20:27:22.494754Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16344","last_updated":"2025-05-31T16:37:32Z","snapshot_observed_at":"2026-08-10T20:19:27.664864Z","submitted_at":"2025-01-15T06:30:17Z","title":"WhiSPA: Semantically and Psychologically Aligned Whisper with Self-Supervised Contrastive and Student-Teacher Learning","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T20:27:22.494754Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2501.16344"},"observation_digest":"sha256:28bca761e709ec953727dd8d44b755ed1af54cb35aa5c38caf9c0909aecae211","observation_id":"dc48ff5f-8d49-4f63-8941-bf4181efd1bc","resolution":{"observed_at":"2026-08-10T20:27:22.494754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-07T14:58:57.191969Z","title":"Preprint, arXiv:2106.07447","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16691","last_updated":"2025-05-23T05:07:17Z","snapshot_observed_at":"2026-08-10T02:50:17.955995Z","submitted_at":"2025-05-22T13:57:02Z","title":"EZ-VC: Easy Zero-shot Any-to-Any Voice Conversion","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T14:58:57.191969Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2505.16691"},"observation_digest":"sha256:128752642b91401d8d450f5fb0fa236740bf04b7d812ffba54e2b1af8c2cebdb","observation_id":"3efc365d-704a-4714-8d27-54e70b726133","resolution":{"observed_at":"2026-08-07T14:58:57.191969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-07T05:20:04.453086Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.08372","last_updated":"2025-06-10T02:37:42Z","snapshot_observed_at":"2026-08-10T11:02:14.049746Z","submitted_at":"2025-06-10T02:37:42Z","title":"Multimodal Zero-Shot Framework for Deepfake Hate Speech Detection in Low-Resource Languages","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:20:04.453086Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.08372"},"observation_digest":"sha256:fde0a4435112f7ae8643727e4a8282e701dd80696519d6b7913e579ee4ab3804","observation_id":"c86a60e8-64fa-429d-b203-491345f93955","resolution":{"observed_at":"2026-08-07T05:20:04.453086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-07T05:26:36.292242Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.11119","last_updated":"2025-06-09T17:52:31Z","snapshot_observed_at":"2026-08-07T05:18:46.575540Z","submitted_at":"2025-06-09T17:52:31Z","title":"Benchmarking Foundation Speech and Language Models for Alzheimer's Disease and Related Dementia Detection from Spontaneous Speech","version":1},"reference_index":3460,"source":"pdf_text","source_observed_at":"2026-08-07T05:26:36.292242Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.11119"},"observation_digest":"sha256:d59a8559fede37f77b1e62a4e2ce5985e2e3af0fce1fbff54586814294eec57f","observation_id":"b907b8c8-5dd0-4009-a434-2028a84c0a45","resolution":{"observed_at":"2026-08-07T05:26:36.292242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T23:41:49.569844Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.17351","last_updated":"2025-06-20T01:28:43Z","snapshot_observed_at":"2026-08-11T07:46:06.032681Z","submitted_at":"2025-06-20T01:28:43Z","title":"Zero-Shot Cognitive Impairment Detection from Speech Using AudioLLM","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:41:49.569844Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.17351"},"observation_digest":"sha256:d706aaa913dccc0a75938203af4020f60611454bd1a8f2bf1607b9cf6628ff9e","observation_id":"2757860d-2e1a-4049-800c-f969eeb485a8","resolution":{"observed_at":"2026-08-06T23:41:49.569844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T21:51:35.040172Z","title":"Shengpeng Ji, Yifu Chen, Minghui Fang, Jialong Zuo, Jingyu Lu, Hanting Wang, Ziyue Jiang, Long Zhou, Shujie Liu, Xize Cheng, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23325","last_updated":"2025-07-09T17:40:35Z","snapshot_observed_at":"2026-08-09T15:45:24.880266Z","submitted_at":"2025-06-29T16:51:50Z","title":"XY-Tokenizer: Mitigating the Semantic-Acoustic Conflict in Low-Bitrate Speech Codecs","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:51:35.040172Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.23325"},"observation_digest":"sha256:a844eaf5c23295f8bb9fc4c2ff99ba6e02380ad9a6429d35d5b5d9554e8991dd","observation_id":"762248e0-df10-4ea3-ac8f-a499ded69049","resolution":{"observed_at":"2026-08-06T21:51:35.040172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T21:36:11.289042Z","title":"Hubert: Self- supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.23869","last_updated":"2025-06-30T14:00:14Z","snapshot_observed_at":"2026-08-09T22:18:43.848984Z","submitted_at":"2025-06-30T14:00:14Z","title":"Scaling Self-Supervised Representation Learning for Symbolic Piano Performance","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:36:11.289042Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.23869"},"observation_digest":"sha256:3e073bdb9d87546968daff92293daf2cc1ff590f4cbd2d55c5e97bad60ee8493","observation_id":"ca1dcac3-5898-4a56-8ae9-a51f4ae5a008","resolution":{"observed_at":"2026-08-06T21:36:11.289042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T22:56:46.224028Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02915","last_updated":"2025-06-25T08:38:27Z","snapshot_observed_at":"2026-08-10T06:26:17.946083Z","submitted_at":"2025-06-25T08:38:27Z","title":"Audio-JEPA: Joint-Embedding Predictive Architecture for Audio Representation Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:56:46.224028Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.02915"},"observation_digest":"sha256:95d100a868d003a474513f6a8961cfd8bf3e601f91aaf27c115fbf65c819003d","observation_id":"2f1a916e-e772-4f3d-8282-a493595b354f","resolution":{"observed_at":"2026-08-06T22:56:46.224028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T19:50:22.355340Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.04554","last_updated":"2025-07-08T12:27:54Z","snapshot_observed_at":"2026-08-08T09:19:22.787731Z","submitted_at":"2025-07-06T22:11:22Z","title":"Self-supervised learning of speech representations with Dutch archival data","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:50:22.355340Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.04554"},"observation_digest":"sha256:e11f67c73987454d69ce5bef3e735bd7f0f439f94ff0df6253ab0b51dd2629b0","observation_id":"ff90399d-1651-42e3-97fb-5e6f8bcc7037","resolution":{"observed_at":"2026-08-06T19:50:22.355340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T15:31:18.775855Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.15641","last_updated":"2025-07-21T14:03:08Z","snapshot_observed_at":"2026-08-10T13:21:16.007541Z","submitted_at":"2025-07-21T14:03:08Z","title":"Leveraging Context for Multimodal Fallacy Classification in Political Debates","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T15:31:18.775855Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.15641"},"observation_digest":"sha256:37e3ae270ad474026f63c877ff489f8ec772f72a9b13c6274e70e886913b03e1","observation_id":"e26faa88-32ee-45c7-8e30-07a804b5bb7d","resolution":{"observed_at":"2026-08-06T15:31:18.775855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:8bbd98f171b00a9c0121732ab85e1e23b4379a04f873516e4af53b5ce7d4e43a","observation_id":"986ec86d-5bf2-489e-a7fd-de4c8516c67d","resolution":{"observed_at":"2026-05-16T05:59:51.132301Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T14:49:52.253608Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.17799","last_updated":"2025-07-23T16:11:44Z","snapshot_observed_at":"2026-08-09T00:03:45.084630Z","submitted_at":"2025-07-23T16:11:44Z","title":"A Concept-based approach to Voice Disorder Detection","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T14:49:52.253608Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.17799"},"observation_digest":"sha256:adeb8ccfb985c474c75bdf5ae8effc64a8e837479a1e9e6d349e55d82bbeea6d","observation_id":"b7d3a70e-6d86-4d32-b58b-f91ce0aac5bf","resolution":{"observed_at":"2026-08-06T14:49:52.253608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T17:33:58.485256Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.16188","last_updated":"2025-08-27T19:49:56Z","snapshot_observed_at":"2026-08-11T04:53:14.351865Z","submitted_at":"2025-08-22T08:08:45Z","title":"Seeing is Believing: Emotion-Aware Audio-Visual Language Modeling for Expressive Speech Generation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T17:33:58.485256Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2508.16188"},"observation_digest":"sha256:d1f8c60d5ee42c8bd0499df0cc60b24f18b2eff5a6c9253c6cc485d43ea975ab","observation_id":"d20f6562-f71c-4c16-8d5f-bdf6236aaf4f","resolution":{"observed_at":"2026-08-05T17:33:58.485256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T13:36:04.730722Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00503","last_updated":"2025-08-30T13:50:58Z","snapshot_observed_at":"2026-08-08T09:17:24.591916Z","submitted_at":"2025-08-30T13:50:58Z","title":"Entropy-based Coarse and Compressed Semantic Speech Representation Learning","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T13:36:04.730722Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2509.00503"},"observation_digest":"sha256:b5a72c61cc50325b96ffed01d3a23d55992317604ee8e62916c082e016cd191b","observation_id":"c9954816-dbc8-43e3-9707-98f3ef9225ec","resolution":{"observed_at":"2026-08-05T13:36:04.730722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-04T13:15:22.832912Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2510.01157","last_updated":"2026-06-09T20:25:55Z","snapshot_observed_at":"2026-08-08T09:22:36.387389Z","submitted_at":"2025-10-01T17:45:04Z","title":"Where Do Backdoors Live? A Component-Level Analysis of Backdoor Propagation in Speech Language Models","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T13:15:22.832912Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2510.01157"},"observation_digest":"sha256:c79a8057cabc92bcef3131ba78ca13a1069e7f5185fdddcd5f291db3694fae1c","observation_id":"a0f2b216-8a76-4676-9c08-b3ca6222aa23","resolution":{"observed_at":"2026-08-04T13:15:22.832912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2603.12221","last_updated":"2026-04-13T08:30:07Z","snapshot_observed_at":"2026-07-06T22:48:53.198077Z","submitted_at":"2026-03-12T17:45:12Z","title":"A Two-Stage Dual-Modality Model for Facial Emotional Expression Recognition","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T11:35:38.330061Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2603.12221"},"observation_digest":"sha256:a22ef08d2da79259a9a29ed06280b0e9ba455496556fc8e2404f2f8738e1a57e","observation_id":"14a60961-3961-41d0-a5f4-b6cea4c81c02","resolution":{"observed_at":"2026-05-15T11:39:59.378823Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-13T17:37:12.659819Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.26292","last_updated":"2026-06-15T19:50:12Z","snapshot_observed_at":"2026-08-08T09:39:10.545296Z","submitted_at":"2026-03-27T11:03:08Z","title":"findsylls: A Language-Agnostic Toolkit for Syllable-Level Speech Tokenization and Embedding","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-13T17:37:12.659819Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2603.26292"},"observation_digest":"sha256:9d7c8c3baa0667ed0e4a2b52eb82742189945315ffa419e968b9b06efe9951a3","observation_id":"baacf565-a834-462f-bf1b-b9e2c8474f3a","resolution":{"observed_at":"2026-07-13T17:37:12.659819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2604.08562","last_updated":"2026-03-17T16:07:15Z","snapshot_observed_at":"2026-07-06T22:57:35.435713Z","submitted_at":"2026-03-17T16:07:15Z","title":"Neural networks for Text-to-Speech evaluation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T09:46:26.884551Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2604.08562"},"observation_digest":"sha256:8c8cd0990519d654d456678c5035bbb264481346e7aa8757b509067ce2caf074","observation_id":"f7728530-8560-4a6b-9fcb-4bc70517722d","resolution":{"observed_at":"2026-05-15T09:49:54.655165Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2605.09152","last_updated":"2026-05-09T20:30:15Z","snapshot_observed_at":"2026-08-11T13:07:02.250319Z","submitted_at":"2026-05-09T20:30:15Z","title":"Meow-Omni 1: A Multimodal Large Language Model for Feline Ethology","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T04:02:23.847187Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2605.09152"},"observation_digest":"sha256:7dba2a969a02c88a4b7014185dd81c880b898d1398bdf1ec368a8a24bfe6cfe5","observation_id":"b1c58a5a-15e7-47fa-849e-a00281d74a84","resolution":{"observed_at":"2026-05-12T06:41:43.777162Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2605.19224","last_updated":"2026-05-19T00:49:51Z","snapshot_observed_at":"2026-07-06T23:29:57.003383Z","submitted_at":"2026-05-19T00:49:51Z","title":"Fine-tuning language encoding models on slow fMRI improves prediction for fast ECoG","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T06:48:52.078640Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2605.19224"},"observation_digest":"sha256:bd8e9ec8d677b56fda2761bc3d0470a0567afeb808e86a45013f3c7503a6cd04","observation_id":"590cc051-77a2-40c2-9753-b2bdf1d71bc4","resolution":{"observed_at":"2026-05-20T06:53:06.014414Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.06357","last_updated":"2026-06-04T16:25:07Z","snapshot_observed_at":"2026-07-06T23:46:10.612537Z","submitted_at":"2026-06-04T16:25:07Z","title":"F3-Tokenizer: Taming Audio Autoencoder Latents for Understanding and Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T23:36:27.369551Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.06357"},"observation_digest":"sha256:3dec6456a202a8b83f52326129e3c8f3a35b9e01b5a6c89d31fd43e2b5e1ec90","observation_id":"27588fde-9bd5-4255-a614-0f6e73963047","resolution":{"observed_at":"2026-07-02T15:47:06.020988Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.11542","last_updated":"2026-06-10T01:07:32Z","snapshot_observed_at":"2026-08-04T09:55:13.088817Z","submitted_at":"2026-06-10T01:07:32Z","title":"Pretrained self-supervised speech models can recognize unseen consonants","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T10:14:47.932613Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.11542"},"observation_digest":"sha256:51f8abccb753c075556c413dbcdce3f08c3142c4caab8f90dc7a8a648b318e67","observation_id":"fcef0e10-8117-47e2-959c-c7bf9d3907e5","resolution":{"observed_at":"2026-07-03T10:07:56.083448Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.19910","last_updated":"2026-06-23T10:40:32Z","snapshot_observed_at":"2026-08-05T15:52:37.159854Z","submitted_at":"2026-06-18T08:04:16Z","title":"Light-weight Pronunciation Assessment via Discrete Speech Token Surprisal","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-26T17:37:25.043607Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.19910"},"observation_digest":"sha256:ad977db90718a3cd21cd17b4308bd4777c0b01425bcc060c0f0e11412437b52e","observation_id":"46238d6e-7817-465c-bc14-d6371a8c32b9","resolution":{"observed_at":"2026-07-04T03:49:30.223989Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.22473","last_updated":"2026-06-21T12:33:44Z","snapshot_observed_at":"2026-07-06T23:57:18.596834Z","submitted_at":"2026-06-21T12:33:44Z","title":"Interleaved Speech Language Models Latently Work In Text","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T10:41:19.777779Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.22473"},"observation_digest":"sha256:81950927062b1335bb2afde56808e9e28f3b49fb90eaba852a10184c4a63cdf3","observation_id":"4cba52db-200a-4e08-9c0a-e31d1b391a8c","resolution":{"observed_at":"2026-07-04T08:59:42.891310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.24910","last_updated":"2026-06-19T08:07:13Z","snapshot_observed_at":"2026-08-06T07:38:22.689273Z","submitted_at":"2026-06-19T08:07:13Z","title":"End-to-End Voice Intent Recognition for Spontaneous Human-Drone Interaction with Naive Users","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-26T13:30:12.101045Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.24910"},"observation_digest":"sha256:d1a6cd1c2ec727986d3bce9cff34610ae5c4163ba0b14d91fd080621bf6c31cf","observation_id":"12d732ad-24e4-49be-9020-97b89026ede6","resolution":{"observed_at":"2026-07-04T07:29:38.221570Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.27206","last_updated":"2026-06-25T16:02:14Z","snapshot_observed_at":"2026-08-02T08:21:01.224967Z","submitted_at":"2026-06-25T16:02:14Z","title":"Syntactic Belief Update as the Driver of Garden Path Processing Difficulty","version":1},"reference_index":288,"source":"arxiv_source","source_observed_at":"2026-06-26T04:38:01.183423Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.27206"},"observation_digest":"sha256:c11b6c848f9a6b00c94ba2d933f04ed8ad2f31845f01c975a47702a1d42aa562","observation_id":"05e6240c-cfc1-42e9-b23f-5ac9f84ef1b8","resolution":{"observed_at":"2026-06-26T04:38:58.619776Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:09a6e3413495aeed000dd58b01c13ea35eb6f0210c7492b89a022a8520a939d6","observation_id":"28de5576-2bb0-49ca-9446-7276fb253d47","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-12T06:06:46.343211Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02920","last_updated":"2026-07-03T03:24:38Z","snapshot_observed_at":"2026-08-09T00:13:42.364327Z","submitted_at":"2026-07-03T03:24:38Z","title":"Layer-wise Cross-Lingual Depression Detection from Speech: Analysis with Contrastive Alignment","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-12T06:06:46.343211Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.02920"},"observation_digest":"sha256:749df826306da0d6fb894e236b2ff74471ecc0d76d853c26c7bd5719be5d54db","observation_id":"2e838a15-a1a9-403d-9d33-4fc1b0085c1f","resolution":{"observed_at":"2026-07-12T06:06:46.343211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2607.06875","last_updated":"2026-07-08T00:17:20Z","snapshot_observed_at":"2026-08-07T17:11:05.338268Z","submitted_at":"2026-07-08T00:17:20Z","title":"Video2Reaction: Mapping Video to Audience Reaction Distribution in the Wild","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-09T23:51:55.422232Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.06875"},"observation_digest":"sha256:07bf50ddca68b6a0e9269021e7c02e1de96a0ad26f8cead3daa62682f4d15b8b","observation_id":"3df35c21-9c57-4186-97af-02c2123f6fe0","resolution":{"observed_at":"2026-07-09T23:56:38.448201Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-01T07:10:39.830211Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.21540","last_updated":"2026-07-24T18:05:20Z","snapshot_observed_at":"2026-08-07T09:16:01.540251Z","submitted_at":"2026-07-23T17:25:08Z","title":"DONDO: Open w2v-BERT Speech-Recognition Base Models for African Languages","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T07:10:39.830211Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.21540"},"observation_digest":"sha256:fd7ea8c755e969b7f811f9a7ad7dcc05b8d3ff9035be6b7cb2e193d5042b9be2","observation_id":"0c6ac7ef-ce87-48c6-acb2-7e2bfde25cf9","resolution":{"observed_at":"2026-08-01T07:10:39.830211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2106.07447/citation-record","integrity":"/paper/2106.07447/integrity","json":"/paper/2106.07447/citation-record.json","paper":"/paper/2106.07447"},"outbound":[],"paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2106.07447."}