{"as_of":"2026-08-09T16:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2724bd57da61ae12a13857bc486877a05b4bf3aaca935207943a8f47f63e4193","coverage":[{"denominator":35,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:17:18.443046Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:17:14.961641Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-01T21:36:15.429731Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-08-04T16:17:14.961641Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:14.961641Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:dd54658a21ef037b0ecab2f6b23f673b8f84770e144dfd16fe736c29b610cdf4","observation_id":"197a8495-d4f5-40e1-b4bf-2d54ab5e1cd0","resolution":{"observed_at":"2026-08-04T16:17:14.961641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":"2509.15001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-07-01T21:36:15.429731Z","title":"Babyhubert: Multilingual self-supervised learning for segmenting speakers in child-centered long-form recordings","venue":"eess.AS","work_id":"4b68d120-3d68-453c-9053-68524cb61b19","year":2025},"citing_paper":{"arxiv_id":"2605.19130","last_updated":"2026-05-18T21:30:54Z","snapshot_observed_at":"2026-08-01T19:57:38.697860Z","submitted_at":"2026-05-18T21:30:54Z","title":"EgoBabyVLM: Benchmarking Cross-Modal Learning from Naturalistic Egocentric Video Data","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T12:12:45.251924Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2605.19130"},"observation_digest":"sha256:bed716677f996c0f7fe8c9cf3830ef9bc2ad345d21a0ad17d2bbaf98571d8532","observation_id":"93e9ca60-a992-4c37-9e85-87fe1977cb74","resolution":{"observed_at":"2026-06-30T03:17:23.577707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":"2509.15001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-07-01T21:36:15.429731Z","title":"Babyhubert: Multilingual self-supervised learning for segmenting speakers in child-centered long-form recordings","venue":"eess.AS","work_id":"4b68d120-3d68-453c-9053-68524cb61b19","year":2025},"citing_paper":{"arxiv_id":"2606.01134","last_updated":"2026-05-31T10:12:47Z","snapshot_observed_at":"2026-08-06T21:55:11.856032Z","submitted_at":"2026-05-31T10:12:47Z","title":"Context-aware child-directed speech detection from long-form recordings","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T16:41:39.242748Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2606.01134"},"observation_digest":"sha256:caafbbeccec8c95c6ec0604592e2bae353108472a5cc45f5b05638dcc01df60d","observation_id":"32db9e29-730a-4637-ade3-0a6bc1b8ff2a","resolution":{"observed_at":"2026-07-01T21:36:15.431349Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-07-12T04:11:03.723969Z","title":"Available: https://arxiv.org/abs/2509.15001","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03201","last_updated":"2026-07-03T11:14:54Z","snapshot_observed_at":"2026-08-09T14:26:17.916824Z","submitted_at":"2026-07-03T11:14:54Z","title":"Deriving Benchmarking Datasets from Long-Form Recordings: Challenges and Opportunities","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-12T04:11:03.723969Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2607.03201"},"observation_digest":"sha256:9bbd09b715ce3bce304e9e0c51031487223a65aacb359539ff2c4b0aa8d940e9","observation_id":"888dcef7-0ea1-4f2f-afb2-793b987d7be0","resolution":{"observed_at":"2026-07-12T04:11:03.723969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.15001/citation-record","integrity":"/paper/2509.15001/integrity","json":"/paper/2509.15001/citation-record.json","paper":"/paper/2509.15001"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:14.871859Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:14.871859Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:712f2256e84b0cf7109d21cc5c6e3048d5eb20ee92cf980b7582e466dc1fa87e","observation_id":"7de6dd56-37e1-4350-8836-46d899941921","resolution":{"observed_at":"2026-08-04T16:17:14.871859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-08-04T16:17:14.961641Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:14.961641Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:dd54658a21ef037b0ecab2f6b23f673b8f84770e144dfd16fe736c29b610cdf4","observation_id":"197a8495-d4f5-40e1-b4bf-2d54ab5e1cd0","resolution":{"observed_at":"2026-08-04T16:17:14.961641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.025234Z","title":"Datasets Our pre-training dataset comprises 19 diverse datasets spanning mul- tiple continents over 40 languages (see Table 1)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.025234Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:7738881d3280ced34d6817709a7882891fbcff5e49e0313a24bd9d852e23723f","observation_id":"face384d-4aea-4164-912d-416f698c3169","resolution":{"observed_at":"2026-08-04T16:17:15.025234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.173475Z","title":"BabyHuBERT-2 achieves 64.0% average F-score, substan- tially outperforming both W2V2-LL4300 (58.7%) and HuBERT base (51.4%)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.173475Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:ab36759324349788d2f7543657c98f944070820c3f4c30ce5431455035926ea0","observation_id":"510ceb85-f257-4c63-b636-4d5d40e6f4d3","resolution":{"observed_at":"2026-08-04T16:17:15.173475Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.251303Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.251303Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:b47469a7744c61115646b31298370abbc596082ad1e4920dd95ce332b97d864f","observation_id":"5b1683b2-2c2b-424a-85c5-e4e61d791237","resolution":{"observed_at":"2026-08-04T16:17:15.251303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.319899Z","title":"ED: ERC (InfantSimulator); AC and TK: ERC (ExELang, 101001095)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.319899Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:7c163fc76156ea15bc3eeae89756176368ee8f54a3fa7645fceb4ae9d968f487","observation_id":"f79fc0c0-83aa-40a2-8a68-5c153a43870c","resolution":{"observed_at":"2026-08-04T16:17:15.319899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04710","last_updated":"2025-03-06T18:57:16Z","snapshot_observed_at":"2026-08-07T17:23:49.018445Z","submitted_at":"2025-03-06T18:57:16Z","title":"Self-Supervised Models for Phoneme Recognition: Applications in Children's Speech for Reading Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04710","snapshot_observed_at":"2026-08-04T16:17:15.909983Z","title":"Self-supervised models for phoneme recognition: Applica- tions in children’s speech for reading learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.909983Z"},"links":{"cited_paper":"/paper/2503.04710","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:9f8961baec2ed51e920f33858aa5cd881cd9aae865043ec7ecf1be1d61834797","observation_id":"7c306f31-f294-46c4-b7e2-7497abf74319","resolution":{"observed_at":"2026-08-04T16:17:15.909983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.375905Z","title":"Long-form recordings to study children’s language input and output in under-resourced contexts,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.375905Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:0ed6bde19a244d8944997a21efabdce39e30705c6e6c4d00df87930457cba9d8","observation_id":"5ebe4f6e-9d68-4e01-9ef2-c39762f82b1d","resolution":{"observed_at":"2026-08-04T16:17:15.375905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11075","last_updated":"2025-06-04T01:45:42Z","snapshot_observed_at":"2026-08-07T10:59:43.042203Z","submitted_at":"2025-06-04T01:45:42Z","title":"Fifteen Years of Child-Centered Long-Form Recordings: Promises, Resources, and Remaining Challenges to Validity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11075","snapshot_observed_at":"2026-08-04T16:17:15.475230Z","title":"Fifteen years of child-centered long-form recordings: Promises, resources, and remaining challenges to validity,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.475230Z"},"links":{"cited_paper":"/paper/2506.11075","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:8960173235c6efad6541d0b4359f490d4fc70b51d7090c875b1c8936ddba97cf","observation_id":"202ab127-9cbc-4509-b055-e5bc040dff58","resolution":{"observed_at":"2026-08-04T16:17:15.475230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.594105Z","title":"Acoustics of children’s speech: Developmental changes of temporal and spectral parameters,","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.594105Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:db8d940a706142e2ee549b5b54648f2498bf8a41a376a3822c323ce118201d31","observation_id":"b8c07514-1220-41b0-a806-cedd3c2526c1","resolution":{"observed_at":"2026-08-04T16:17:15.594105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.686273Z","title":"Acous- tic variability and automatic recognition of children’s speech,","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.686273Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:cd3ae871a1838a7c229b45d0d4e281bad8a98411239188bb5e486f0dc0b8685d","observation_id":"4fce2812-16b9-47ff-a12f-dd9e1ec3d8cb","resolution":{"observed_at":"2026-08-04T16:17:15.686273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.06733","last_updated":"2021-10-13T14:03:07Z","snapshot_observed_at":"2026-08-02T04:37:32.946853Z","submitted_at":"2021-10-13T14:03:07Z","title":"Systematic Inequalities in Language Technology Performance across the World's Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.06733","snapshot_observed_at":"2026-08-04T16:17:15.777749Z","title":"Systematic inequalities in language technology performance across the world’s languages,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.777749Z"},"links":{"cited_paper":"/paper/2110.06733","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:522f6d8309ad439513bcecb9c7069de3289c8990cb9becf50b8dcb7b6f4c96a7","observation_id":"0300132c-f066-434f-b5b7-a68eba132b49","resolution":{"observed_at":"2026-08-04T16:17:15.777749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.846220Z","title":"Introduction To Partial Fine-tuning: A Comprehensive Evaluation Of End-to-end Chil- dren’s Automatic Speech Recognition Adaptation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.846220Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:28f8b9c84c67fe25ad4841e321966d3b098bdfe26247c45ba2ebcf8be2373a1d","observation_id":"988e1431-a2c6-4edc-8a5e-6a8d7b2284f9","resolution":{"observed_at":"2026-08-04T16:17:15.846220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.645092Z","title":"A thorough eval- uation of the language environment analysis (lena) system,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.645092Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:7a907304f7006e48f8f82769865bf14ec74ef8f8e33776c228d1a8624cf2104a","observation_id":"08868511-8a62-416a-969a-e319ae95028c","resolution":{"observed_at":"2026-08-04T16:17:16.645092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.065134Z","title":"Towards robust family-infant audio analysis based on unsu- pervised pretraining of wav2vec 2.0 on large-scale unlabeled family audio,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.065134Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:664abe92aa60c89ce175576756cec2555377d8b8230363782aa830d264c8715f","observation_id":"dfa272ec-0ca2-46d5-aed5-fba243dff1c1","resolution":{"observed_at":"2026-08-04T16:17:16.065134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.163714Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representa- tions,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.163714Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:dae1b6fc981ad97659c2e63ebc806ffdf496b5e45276fe65fe788efeed61a3df","observation_id":"feb48427-a221-43cd-8f5f-8576c170fbf7","resolution":{"observed_at":"2026-08-04T16:17:16.163714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.231220Z","title":"Employing self- supervised learning models for cross-linguistic child speech maturity classification,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.231220Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:773c4fd7475474c931ee64c153c20e65b7e9795d9922b518f9719324c3b41a5b","observation_id":"029aa4a9-e35c-4fa9-8e01-6fa19d4b96bb","resolution":{"observed_at":"2026-08-04T16:17:16.231220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.293783Z","title":"Reverse en- gineering language acquisition with child-centered long-form recordings,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.293783Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:9e5e77f5ff1049bd7990d985a6aed7c8fc869ee0337262901854b5a686bcb1d4","observation_id":"f172a12f-aa0d-43f6-9edf-0b122c1bb1eb","resolution":{"observed_at":"2026-08-04T16:17:16.293783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.386578Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.386578Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:6498182d9e7dffb7d4572db4389c1c44a84b7e25ef88221b63b2eb69d54d258d","observation_id":"a1b580f4-1acc-4e78-97cf-280634a1b5f1","resolution":{"observed_at":"2026-08-04T16:17:16.386578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.495794Z","title":"Homebank: An online repository of daylong child-centered audio recordings,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.495794Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:1804388dba4a39eb3ee67f7c2f6ad55882fce0f43d98c363449cd8f9dd3782a5","observation_id":"55a9fde8-e000-4f5a-afb9-a854368c5971","resolution":{"observed_at":"2026-08-04T16:17:16.495794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.549676Z","title":"mhubert-147: A compact multilingual hubert model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.549676Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:c1233c85fcba92ca667d929143cc5c0f0113a58af225fa721c5fbfe17c4fa68d","observation_id":"8107da7c-5f4f-41db-9645-9e8d97f94e3e","resolution":{"observed_at":"2026-08-04T16:17:17.549676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.089461Z","title":"For BabyHuBERT-1, we extract features from the 6th layer of WavLM-base-plus","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.089461Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:7e48f90377ded3766f6348f9255cd750f249229f8f2da8d57fdbd8b2a6de3677","observation_id":"48f06c87-4cdd-4cb6-81b1-981995ff4aaa","resolution":{"observed_at":"2026-08-04T16:17:15.089461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.766857Z","title":"An open-source voice type classifier for child-centered daylong recordings,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.766857Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:5d4568f8bb7a3d997ec5e88156eec9f6b38509b5fd3f0942fb774a283ef8897f","observation_id":"c04219f6-2123-4d13-b872-af115a88c0bb","resolution":{"observed_at":"2026-08-04T16:17:16.766857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.859633Z","title":"Signal processing for young child speech language development.,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.859633Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:8640344e620b5582be7ecafd7623604afe2f924da1ecdf10a647a18fefec6aa1","observation_id":"22b0a1c0-bf6b-4fda-91a7-6d7c4f6bed64","resolution":{"observed_at":"2026-08-04T16:17:16.859633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.009526Z","title":"Speaker recognition from raw waveform with sincnet,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.009526Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:2d6315fbebbb8b31a619b9fc0da9b1d52325954713f1d8b67af55c5c728f5865","observation_id":"bf2eca15-1ff9-4885-ba3a-68295aecf0e3","resolution":{"observed_at":"2026-08-04T16:17:17.009526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.129113Z","title":"Challenges in Automated Processing of Speech from Child Wearables: The Case of V oice Type Classifier,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.129113Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:3f333b415e2f76a4c48914310db20f9f5c372820d227fec1ff45f3eb02f7c37a","observation_id":"65504122-2325-4306-89d8-5b34d2807105","resolution":{"observed_at":"2026-08-04T16:17:17.129113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.275963Z","title":"Developing a cross-cultural annotation system and metacorpus for studying infants’ real world language experience,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.275963Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:51a95cec5ad3b959b8cf268a93306ded1e24f4dfd156044a86e3ed3099ef9d64","observation_id":"e5ff6c42-0ded-4034-9da0-f17e535428c1","resolution":{"observed_at":"2026-08-04T16:17:17.275963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.359917Z","title":"Wavlm: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.359917Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:7fb96104f29d450048866df15fa592ab7bb607a985c9bf3dc0fd80dfedb0da0d","observation_id":"55f4bc48-c8bd-4414-895d-46353011723e","resolution":{"observed_at":"2026-08-04T16:17:17.359917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.785559Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hid- den units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.785559Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:c1bb8fcbec06ee7e42bb47e8d34c0fde0f0d0180c97896677a95fd269b1b21f2","observation_id":"16dfa19a-e731-410e-bb84-cf5f3b0f51f1","resolution":{"observed_at":"2026-08-04T16:17:17.785559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.961650Z","title":"Superb: Speech processing universal performance benchmark,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.961650Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:30bc10b8c44ac29ccfb5a79e4bc654e945401830e1317b2584f00cdf8fe67a5f","observation_id":"cfe825c6-7bba-418e-8b32-d845b1e1b0be","resolution":{"observed_at":"2026-08-04T16:17:17.961650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.159359Z","title":"pyannote. metrics: A toolkit for reproducible evaluation, diagnostic, and error analysis of speaker diarization systems.,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.159359Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:778034c467209e3151e7863350751aa71738a700c60bcc4634789a8c8b80d013","observation_id":"dfe15651-fce5-4a50-9d6b-31680f076f1f","resolution":{"observed_at":"2026-08-04T16:17:18.159359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.216118Z","title":"Torchaudio 2.1: Advancing speech recognition, self-supervised learning, and audio process- ing components for pytorch,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.216118Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:596d3f5e2bc509166bea66623276652cacac1c152f0e532fc4e7d2be739b1fa4","observation_id":"7cdeb22e-d547-4360-b9d4-1469b816b4a0","resolution":{"observed_at":"2026-08-04T16:17:18.216118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.319135Z","title":"Scikit-learn: Machine learning in Python,","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.319135Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:d860589709c8e0c39ffc2c5e1b0a466f4b4b344afacbafff5154eccc555fe7be","observation_id":"5ac68a9b-1902-4b7b-8171-71b48a6242cb","resolution":{"observed_at":"2026-08-04T16:17:18.319135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.361461Z","title":"Child-directed and overheard input from different speakers in two distinct cultures,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.361461Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:c72edba71a0d661ef596b553481e4c2bd10b0605a0970872b8902297a628d7df","observation_id":"6e801bbe-5292-40a0-b279-9a4ff93a2c25","resolution":{"observed_at":"2026-08-04T16:17:18.361461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.443046Z","title":"Putting the child in the driver’s seat: insights into language development from children’s interactions in preschool classrooms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.443046Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:e75697769288d9229ed356ba731d16d070508ef9ae5fdfce1fa90ba1d263acd3","observation_id":"f6fc1119-c1c3-47ca-9202-f2741eb1efff","resolution":{"observed_at":"2026-08-04T16:17:18.443046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-05T23:26:08.097511Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings"},"reference_resolution":{"displayed":35,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":34,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":35},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 35 of 35 outbound references and 4 inbound Pith citation observations for arXiv:2509.15001."}