{"as_of":"2026-08-10T07:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e8e121ec98fff4ece69f6187bdae10b476247b7f048bbb909e49df83d163f6bc","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-08T07:33:22.898615Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-08T07:33:22.898615Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T07:34:43.130062Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"cited_work":{"arxiv_id":"2607.06392","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.06392","snapshot_observed_at":"2026-07-08T07:34:43.130062Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","venue":"cs.SD","work_id":"93fceddf-4119-4963-a7fa-016a6c5d3bd1","year":2026},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2607.06392","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:e10c182fe05fb3a451a550c37052839326b1500d32c9f007f9d48f296b54bda3","observation_id":"82848d1c-5f7d-4275-8eaa-c256a288cd70","resolution":{"observed_at":"2026-07-08T07:34:43.131579Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2607.06392/citation-record","integrity":"/paper/2607.06392/integrity","json":"/paper/2607.06392/citation-record.json","paper":"/paper/2607.06392"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.372864Z","title":null,"venue":null,"work_id":"9ec27f02-5dca-4216-b734-823b200b1b35","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:e2a3b85206b0967bb2ded5bd80d80ad3a84744cecf34e8e779c97eaa60335508","observation_id":"5e6941ad-25da-4c07-acd2-abeff45c5e65","resolution":{"observed_at":"2026-07-08T07:34:43.374441Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"cited_work":{"arxiv_id":"2607.06392","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.06392","snapshot_observed_at":"2026-07-08T07:34:43.130062Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","venue":"cs.SD","work_id":"93fceddf-4119-4963-a7fa-016a6c5d3bd1","year":2026},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2607.06392","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:e10c182fe05fb3a451a550c37052839326b1500d32c9f007f9d48f296b54bda3","observation_id":"82848d1c-5f7d-4275-8eaa-c256a288cd70","resolution":{"observed_at":"2026-07-08T07:34:43.131579Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.307113Z","title":"P” for predictive (masked prediction), “C","venue":null,"work_id":"257127dd-3bc6-448f-b278-d7f1ff4db3ca","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9828454e95520f84ed3808996299d6b89b220ea6930ad0d92d3783784be537f8","observation_id":"ed0d7673-27ed-4ba9-bc35-6a55de3b41b2","resolution":{"observed_at":"2026-07-08T07:34:43.309303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.387703Z","title":"flattens","venue":null,"work_id":"418e6a90-bf28-4583-98fb-7701a48de5bf","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:b574b7100a12f824638f7a492abfc3e07e989c5e2a5fdc696c6c760c89f94830","observation_id":"2de6934a-ee2a-4bbd-9f78-5582d959bb2e","resolution":{"observed_at":"2026-07-08T07:34:43.389772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.339658Z","title":"In terms of phonetic content (SpeechBERTScore), both SSL models display a broad region of high SpeechBERTScore values, indicating a stable phonetic encoding","venue":null,"work_id":"c0047eb9-c4a9-4a32-b1a7-b45b70ec0f58","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:704f5b1b799117a75f2be4c17c15f273051e853c5cb5f357d198e6222359aab3","observation_id":"77fa1bdf-2c72-49b1-b82b-e0aa07855210","resolution":{"observed_at":"2026-07-08T07:34:43.342170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.343227Z","title":"Our analysis revealed dis- tinct optimization regimes, notably the late-stageentropy col- lapsein Wav2Vec2, contrasting with the geometric stability of WavLM","venue":null,"work_id":"19513fbc-0d34-4aef-8fcf-af913c61d088","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:3545db15a2273a0fd55ca92c238b91bd930686cf456277b6ae3e0cf74c7c7b05","observation_id":"9bec8b19-df44-4e57-96ff-b88bf026c7d0","resolution":{"observed_at":"2026-07-08T07:34:43.347145Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.319279Z","title":null,"venue":null,"work_id":"b2cc2478-1ca8-4bb5-833e-bf496dfec392","year":2025},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:96c6bc966812a2ded0130f3eec687bc169fd6373d292b4fb17cde6ab4b5ebfd9","observation_id":"27b71982-218e-42f6-a1a0-761dd69a6a5e","resolution":{"observed_at":"2026-07-08T07:34:43.322052Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.293105Z","title":"They did not contribute to the scientific content, analyses, or conclusions of this work","venue":null,"work_id":"b4f9750b-0861-438d-9c7c-1325dc7932d3","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:43c76e11ea722ea7caa9629106e05415d94a9972ecba87c317eea5850a239274","observation_id":"35cb1050-b732-42c7-9696-4498b151b0d0","resolution":{"observed_at":"2026-07-08T07:34:43.295532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.301897Z","title":"Audio self-supervised learning: A survey","venue":null,"work_id":"bdebf7c0-605c-40a4-8309-10c8cd5e56c9","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:48eb84ee085691efd3f2eb4628093958b4481315f3fcc0f984a93580b04c0998","observation_id":"8e70f6de-0402-4dbd-8dd8-12f0d27e7eaf","resolution":{"observed_at":"2026-07-08T07:34:43.303685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.296267Z","title":"Self-supervised speech representation learning: A review","venue":null,"work_id":"2490870d-7d5c-4794-a031-fd48af7be202","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:daf88074113ba8996b0d846c43416329ab665757a5fdbe7f6ad92f52de9f6f2b","observation_id":"dfbe4b0c-e9ee-4e7e-9929-612bc688dc09","resolution":{"observed_at":"2026-07-08T07:34:43.298028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T12:16:14.563919Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing","venue":null,"work_id":"1d47d864-93dd-41e6-9426-cfc627321f9f","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:81aaed6075a10df739a2a21a29fb1861dca6cc33af17c75a6353403c84cd95b4","observation_id":"bbe2d59d-b28a-451d-90d9-74bb0d110a6d","resolution":{"observed_at":"2026-07-08T07:34:43.335659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.393644Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech repre- sentations","venue":null,"work_id":"a3ebe5eb-bb98-4850-b76b-36c4083e687e","year":2020},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:3ced0a22077b6a7ae2a7ca1fe5d7a0ed33851a4664c8337c4b1c4eaea29b3437","observation_id":"b5720b33-bfa1-429f-8939-75636440af77","resolution":{"observed_at":"2026-07-08T07:34:43.396092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.404293Z","title":"Hubert: Self-supervised speech represen- tation learning by masked prediction of hidden units","venue":null,"work_id":"5e9a152d-1afa-42b6-b2a2-c022f7fce72a","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:fc1f0451e0f1c66a5d162c33b050d0e51cbd5e8a87d353fb4f975ce3ee0bce39","observation_id":"b2204259-f1a5-4f4b-88be-e31f10c30167","resolution":{"observed_at":"2026-07-08T07:34:43.406451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.06185","last_updated":"2021-01-14T14:17:22Z","snapshot_observed_at":"2026-08-05T23:11:57.856234Z","submitted_at":"2020-12-11T08:22:23Z","title":"Exploring wav2vec 2.0 on speaker verification and language identification","version":2},"cited_work":{"arxiv_id":"2012.06185","doi":null,"metadata_source":"pith","pith_arxiv_id":"2012.06185","snapshot_observed_at":"2026-07-08T07:34:43.107921Z","title":"Exploring wav2vec 2.0 on speaker verification and language identification","venue":"cs.SD","work_id":"a5657a14-7f45-4606-9769-988459cbd727","year":2020},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2012.06185","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:fb3d09dc6b42bc377b902fd4a43c994ec0feef448a1aec01a810ebcac8f4dbc4","observation_id":"a363e606-192b-4193-9151-e4a170cc95c5","resolution":{"observed_at":"2026-07-08T07:34:43.109415Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02735","last_updated":"2022-10-03T20:50:54Z","snapshot_observed_at":"2026-07-06T12:05:22.076764Z","submitted_at":"2021-11-04T10:39:06Z","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","version":3},"cited_work":{"arxiv_id":"2111.02735","doi":null,"metadata_source":"pith","pith_arxiv_id":"2111.02735","snapshot_observed_at":"2026-07-11T01:57:52.037436Z","title":"A fine-tuned wav2vec 2.0/hubert benchmark for speech emotion recognition, speaker verification and spoken language understanding","venue":"cs.CL","work_id":"6bb9f860-e45b-4ceb-8ca3-17d38124e800","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2111.02735","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:abe2cc15916b2461fd4bfa200c4e477700c4497235e192a21a059fa5aecf83c9","observation_id":"5890da5e-07d5-4cb0-a070-6e64d930fa15","resolution":{"observed_at":"2026-07-08T07:34:43.128546Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03502","last_updated":"2021-04-08T04:31:58Z","snapshot_observed_at":"2026-07-06T10:57:30.968546Z","submitted_at":"2021-04-08T04:31:58Z","title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","version":1},"cited_work":{"arxiv_id":"2104.03502","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.03502","snapshot_observed_at":"2026-07-08T07:34:43.114766Z","title":"Emo- tion recognition from speech using wav2vec 2.0 embeddings","venue":"cs.SD","work_id":"ebe96864-b6a5-48a9-9eee-f0502736a27f","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2104.03502","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:81c92e24e8c3dbd422c4f9fd38fb11b5ca4c58b9c1355578632e421a8c936ae6","observation_id":"d94f2228-de5b-4331-8f5c-64da30fd43e3","resolution":{"observed_at":"2026-07-08T07:34:43.117158Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.377871Z","title":"Exploring wav2vec 2.0 fine tuning for improved speech emotion recognition,","venue":null,"work_id":"40a9f41a-ea94-407d-89a1-01806a5cbb37","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:94b19b6948dcd54f27b1e1884550d1805052e0d48daee2907191fac7da2f2a8b","observation_id":"f4d1bfe8-a277-4c42-ae3b-018a26f983d3","resolution":{"observed_at":"2026-07-08T07:34:43.379822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.03339","last_updated":"2022-07-05T12:30:18Z","snapshot_observed_at":"2026-07-06T12:57:47.412026Z","submitted_at":"2022-04-07T10:22:26Z","title":"Boosting Self-Supervised Embeddings for Speech Enhancement","version":2},"cited_work":{"arxiv_id":"2204.03339","doi":null,"metadata_source":"pith","pith_arxiv_id":"2204.03339","snapshot_observed_at":"2026-07-08T07:34:43.133145Z","title":"Boosting Self-Supervised Embeddings for Speech Enhancement","venue":"eess.AS","work_id":"f1ce3a09-0b6c-460c-99a8-cb9087f20387","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2204.03339","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9fac187ea415e8e5083d5cbd32665e9ad1315ba3807831553a80b4dbb45a2f7e","observation_id":"942f4d2c-fab2-441b-b029-9e908a01ffef","resolution":{"observed_at":"2026-07-08T07:34:43.134667Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.326100Z","title":"Investigating self-supervised learning for speech enhancement and separation,","venue":null,"work_id":"a74e10e3-2660-4472-914c-973d9035dcac","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:47ca6df37b35365404d6f253ecdb97a278d76f0be9b8132fba570ca4d5af7381","observation_id":"dba035b9-43c3-4588-9e90-a248acc7674e","resolution":{"observed_at":"2026-07-08T07:34:43.328172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.310093Z","title":"Layer-wise analysis of a self-supervised speech representation model","venue":null,"work_id":"ffc62a6a-ca4d-4de1-a0b4-181aaffb684c","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:d413e8915d26c8487889c6e7a16e38e1010df2110a564aacae8fa62352510bd5","observation_id":"29d72771-7895-46ba-9604-4f6e0be24cc6","resolution":{"observed_at":"2026-07-08T07:34:43.311979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00387","last_updated":"2021-07-12T22:46:37Z","snapshot_observed_at":"2026-08-07T20:40:16.363405Z","submitted_at":"2021-01-02T06:29:12Z","title":"What all do audio transformer models hear? Probing Acoustic Representations for Language Delivery and its Structure","version":2},"cited_work":{"arxiv_id":"2101.00387","doi":null,"metadata_source":"pith","pith_arxiv_id":"2101.00387","snapshot_observed_at":"2026-07-08T07:34:43.100777Z","title":"What all do audio transformer models hear? probing acoustic rep- resentations for language delivery and its structure","venue":"cs.CL","work_id":"2dd54a11-e44d-4b89-913f-ca5c09c31009","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2101.00387","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:989705a8b40646e28732957c607ddba6f783a94f348007160afdb7ac05fa0c45","observation_id":"36d8a162-c6e8-4401-98cf-c998e26c91d3","resolution":{"observed_at":"2026-07-08T07:34:43.103019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.299014Z","title":"Phone and speaker spatial organization in self-supervised speech represen- tations,","venue":null,"work_id":"53bbe33f-b213-4d47-a8f4-61d3f51034a6","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:a55b8c75d9eb68d0b04dfcf6041f52d22c97831cf11090c7861faef12fda04fb","observation_id":"7c469199-a373-4f13-8db3-5d484cd9d45f","resolution":{"observed_at":"2026-07-08T07:34:43.301070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.383869Z","title":"Exploration of a self- supervised speech model: A study on emotional corpora,","venue":null,"work_id":"2c26a125-14de-427c-8e23-f78d1b9a65e5","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:436691fb01f87a31ce85e802c947dfd121d4f8ed9892cec393932c2adbe7c902","observation_id":"813eae8a-21ba-4a2f-b6f2-aa9d57a6c14d","resolution":{"observed_at":"2026-07-08T07:34:43.386830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.401883Z","title":"What do self-supervised speech and speaker mod- els learn? new findings from a cross model layer-wise analysis,","venue":null,"work_id":"1d09a53c-348b-4b4a-b2bc-a9d0e2829407","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:da329c1b4525a62d7f07e1598dea3c5bfd810c8727abd7e9ba033c4987a19390","observation_id":"8fbb7544-e92a-44e3-8ddf-d03abc8b4269","resolution":{"observed_at":"2026-07-08T07:34:43.403501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.397202Z","title":"Comparative layer-wise anal- ysis of self-supervised speech models,","venue":null,"work_id":"d3e629f2-aa10-4fb8-a634-93d3ae24ffe8","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:22644d59deabfa50d3822a894a4d52caf83bce96e954cf6bc83d8697be3ab571","observation_id":"795b1293-8c35-481f-ae1f-491ace693320","resolution":{"observed_at":"2026-07-08T07:34:43.399306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2501.05310","doi":"10.48550/arxiv.2501.05310","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A large-scale probing analysis of speaker-specific at- tributes in self-supervised speech representations,","venue":"ArXiv.org","work_id":"d3f2fac3-7f36-4861-b473-562f66ebf28c","year":2025},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:2335db8d0ffd580734e102c845d2f90be65e39eab102e91a9792a0baaec61195","observation_id":"06e38de1-db96-468e-a588-fec3d33cf9dd","resolution":{"observed_at":"2026-07-08T07:34:43.099226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02013","last_updated":"2025-06-15T18:27:17Z","snapshot_observed_at":"2026-08-01T16:50:50.645857Z","submitted_at":"2025-02-04T05:03:42Z","title":"Layer by Layer: Uncovering Hidden Representations in Language Models","version":2},"cited_work":{"arxiv_id":"2502.02013","doi":"10.48550/arxiv.2502.02013","metadata_source":"pith","pith_arxiv_id":"2502.02013","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Layer by Layer: Uncovering Hidden Representations in Language Models","venue":"cs.LG","work_id":"7b4ac06a-e804-4f0a-8305-c45f2735afb5","year":2025},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2502.02013","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:7552ed49845362fd95af4688d1a44ad294d31b2af2da9710d16667b641c3e7f5","observation_id":"75a8d211-9522-437a-96ad-26166bbd215e","resolution":{"observed_at":"2026-07-08T07:34:43.095089Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"physics/0004057","last_updated":"2000-04-24T15:22:30Z","snapshot_observed_at":"2026-08-07T10:46:08.274157Z","submitted_at":"2000-04-24T15:22:30Z","title":"The information bottleneck method","version":1},"cited_work":{"arxiv_id":"physics/0004057","doi":"10.48550/arxiv.physics/0004057","metadata_source":"pith","pith_arxiv_id":"physics/0004057","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The information bottleneck method","venue":"physics.data-an","work_id":"72655a80-0724-45ad-a330-1f4ed7aa613b","year":2000},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/physics/0004057","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:2083d1c2769761c70d47d62ff6ec396a3b628f977921c2586987b50e22151c57","observation_id":"6e844674-e6a0-4843-9c64-ee80b6a53b5c","resolution":{"observed_at":"2026-07-08T07:34:43.091656Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-10T23:49:06.793856+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T23:49:06.793856+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1703.00810","last_updated":"2017-04-29T17:32:47Z","snapshot_observed_at":"2026-08-08T16:36:13.828301Z","submitted_at":"2017-03-02T14:53:14Z","title":"Opening the Black Box of Deep Neural Networks via Information","version":3},"cited_work":{"arxiv_id":"1703.00810","doi":"10.48550/arxiv.1703.00810","metadata_source":"pith","pith_arxiv_id":"1703.00810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Opening the Black Box of Deep Neural Networks via Information","venue":"cs.LG","work_id":"3b14f412-2206-469d-bfd3-c387c75ea711","year":2017},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/1703.00810","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:3b39111aee8e19cfe68cd0e0cf9ba510d54f436d869c6de56d3a49adc7d7dec2","observation_id":"b23f0263-c011-4454-b4c2-d61de8dd7a7b","resolution":{"observed_at":"2026-07-08T07:34:43.106484Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-11T19:50:18.979424+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T19:50:18.979424+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.407415Z","title":"To compress or not to compress—self-supervised learning and information theory: A review,","venue":null,"work_id":"fe601810-66ee-493b-9e1f-9d606af72802","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:fbdb0a6df1323843c46fc6814da5d5dd7c558d38ee1374c5b17354327721944f","observation_id":"a152c60a-aefe-49da-a0c1-d8296a74b34d","resolution":{"observed_at":"2026-07-08T07:34:43.409306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.348159Z","title":"Large language models implic- itly learn to straighten neural sentence trajectories to construct a predictive representation of natural language","venue":null,"work_id":"0b7fd58b-0346-4ccd-bec4-ee1f9f6524ce","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:81b3abb2ae8e9df8f6951780211667a1dba9ac0cfd9a46564d873110ecd8a005","observation_id":"cdc9bef5-c658-48ea-86a5-33b26788c949","resolution":{"observed_at":"2026-07-08T07:34:43.350133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":"1807.03748","doi":"10.1609/aaai.v36i10.21390","metadata_source":"pith","pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Representation Learning with Contrastive Predictive Coding","venue":"cs.LG","work_id":"7b08a1d4-d565-424e-9c86-6ef244b7b90a","year":2018},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:ef1003d484aad87885945ca1d7aa939ce227445aeac941ee6d4acd31a3f72b43","observation_id":"48ddab6f-1fa6-47d6-80ea-9b1d26417afc","resolution":{"observed_at":"2026-07-08T07:34:43.085386Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04000","last_updated":"2023-12-07T02:31:28Z","snapshot_observed_at":"2026-08-01T16:37:51.615090Z","submitted_at":"2023-12-07T02:31:28Z","title":"LiDAR: Sensing Linear Probing Performance in Joint Embedding SSL Architectures","version":1},"cited_work":{"arxiv_id":"2312.04000","doi":"10.48550/arxiv.2312.04000","metadata_source":"pith","pith_arxiv_id":"2312.04000","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2312.04000 , year=","venue":"cs.LG","work_id":"3f4999db-6f36-4b5d-a91c-d07f2c21a2ee","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2312.04000","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:97c5024c837884b3f51ec8343302096cd4c3642d1611f332e28b77f2e3edc73d","observation_id":"38a4f022-4b99-4fb5-9bf3-890ef975b247","resolution":{"observed_at":"2026-07-08T07:34:43.125210Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.08164","last_updated":"2023-07-27T18:26:45Z","snapshot_observed_at":"2026-08-06T18:44:25.708968Z","submitted_at":"2023-01-19T16:56:21Z","title":"DiME: Maximizing Mutual Information by a Difference of Matrix-Based Entropies","version":3},"cited_work":{"arxiv_id":"2301.08164","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.08164","snapshot_observed_at":"2026-07-08T07:34:43.110807Z","title":"arXiv preprint arXiv:2301.08164 , year=","venue":"cs.LG","work_id":"2645d62b-e608-4019-b93e-c8fe50ba552d","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2301.08164","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:09005bc6f4bcfc2e785d11e84a336b75493597d8f942b32f2674b7042c9b76a5","observation_id":"2c331c9a-a757-4979-8678-2bdeb9a338a5","resolution":{"observed_at":"2026-07-08T07:34:43.112949Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.369308Z","title":"Multivari- ate extension of matrix-based r ´enyi’s alpha-order entropy func- tional,","venue":null,"work_id":"6f08b0e1-7005-4e8a-9809-152de99d8ea0","year":2019},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:f24735c9b2944d9017a31425f5dee6bcc8323af8f065324548c9f94b8b470fa2","observation_id":"120c595d-e87d-4669-b885-91e85b5d808e","resolution":{"observed_at":"2026-07-08T07:34:43.372137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.356875Z","title":"Data2vec: A general framework for self-supervised learning in speech, vision and language,","venue":null,"work_id":"0cc494b7-0332-4cf3-91e3-354527645f16","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:06e563cd986585a7436abe6ad4c6da238d658d5a0a1bf3e5a6a0c5b76b4f945e","observation_id":"e9e23925-46c6-4ea7-9ffe-5215a8b695ff","resolution":{"observed_at":"2026-07-08T07:34:43.362731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.304509Z","title":"Lib- rispeech: an asr corpus based on public domain audio books","venue":null,"work_id":"56dd2083-75da-478e-a91d-6aef8a2937a4","year":2015},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:4a53ad43ed81487e9c529d7ea8bc3a1dc41bafaeba44282ceea31adb31c9024e","observation_id":"3f48d494-06d2-40d1-b856-621afba09e87","resolution":{"observed_at":"2026-07-08T07:34:43.306218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02747","last_updated":"2023-02-08T15:46:05Z","snapshot_observed_at":"2026-08-02T18:24:58.914589Z","submitted_at":"2022-10-06T08:32:20Z","title":"Flow Matching for Generative Modeling","version":2},"cited_work":{"arxiv_id":"2210.02747","doi":"10.1038/s41467-024-47656-z","metadata_source":"pith","pith_arxiv_id":"2210.02747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flow Matching for Generative Modeling","venue":"cs.LG","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2210.02747","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:5c4931ecc5d5ce039e9a083848a55369df369f88dc879f3d955269668c880045","observation_id":"62b502cf-d6d2-4084-80c8-278417492d17","resolution":{"observed_at":"2026-07-08T07:34:43.121432Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.363806Z","title":"Scalable diffusion models with transform- ers","venue":null,"work_id":"30e140cc-f93d-4815-9180-92379f790636","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:94d3f1b71552bed1fffd0247005a24b874a0c5d9a528aa656509eafb091fa8c1","observation_id":"89897661-0cd8-492d-933a-242e58106ef7","resolution":{"observed_at":"2026-07-08T07:34:43.365662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.336708Z","title":"Hifi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"a0b0841b-3ae3-4344-8bb9-c83e91447e68","year":2020},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:07cfb8d9b881f339cbe4e67ebe175c3001625fc1b5457a4af5fa60bf8670c49e","observation_id":"19569968-b82b-4ed5-83a7-388dc37522a7","resolution":{"observed_at":"2026-07-08T07:34:43.338700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16812","last_updated":"2024-09-01T14:34:27Z","snapshot_observed_at":"2026-08-06T17:12:30.832433Z","submitted_at":"2024-01-30T08:26:28Z","title":"SpeechBERTScore: Reference-Aware Automatic Evaluation of Speech Generation Leveraging NLP Evaluation Metrics","version":3},"cited_work":{"arxiv_id":"2401.16812","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.16812","snapshot_observed_at":"2026-07-08T07:34:43.086784Z","title":"SpeechBERTScore: Reference-aware automatic evaluation of speech generation leveraging nlp evaluation metrics","venue":"cs.SD","work_id":"8d315452-24ee-4b4a-befa-d194323ff285","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2401.16812","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:bafbcb428804f0c848cd22ab3eddc8e01e575568a62c228680d848b362c8872a","observation_id":"69b0a9c6-e15d-42b4-901c-aa49d03539e9","resolution":{"observed_at":"2026-07-08T07:34:43.088448Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.312848Z","title":"Generalized end- to-end loss for speaker verification,","venue":null,"work_id":"09d3ea9b-8847-4901-8d91-ab2564c8e681","year":2018},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:c74a29f97408ba3b99eb9a0e4baa19b150b08ea1f371b9b23c25d1cfc1ae8035","observation_id":"01c1c60e-7173-4b4b-9790-0160e0efcb2e","resolution":{"observed_at":"2026-07-08T07:34:43.317506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.375243Z","title":"High-fidelity neural phonetic posteriorgrams,","venue":null,"work_id":"1379fb12-f3e5-40c4-b95c-5c360375d986","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:082992700372931fea8832696e172caff084710bb5f6f21f2e6a44211089a217","observation_id":"b5c06f6a-c24f-4f81-8d89-b89ba7dfbb1b","resolution":{"observed_at":"2026-07-08T07:34:43.376922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.366713Z","title":"Crepe: A convo- lutional representation for pitch estimation,","venue":null,"work_id":"f2c16c71-198a-4501-bb40-e6c5764a7d15","year":2018},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:e25824d4269383e8664fb47a2e530f192aa5aeab1454b129b3ff6eebd2cdbbf3","observation_id":"e1ba50e0-15ff-4b64-ba78-c9419ee83a97","resolution":{"observed_at":"2026-07-08T07:34:43.368407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":2,"verified_exact":13,"verified_fuzzy":27},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 1 inbound Pith citation observation for arXiv:2607.06392."}