{"as_of":"2026-08-08T11:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d4ab197c243f3b814c426aa5d1a85ead646eb079a209cf0bac9453c7d13682f8","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:08.869391Z","state":"measured"},{"denominator":62,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":62,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:02.787698Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T13:35:26.587502Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02584","snapshot_observed_at":"2026-08-07T11:26:02.787698Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:02.787698Z"},"links":{"cited_paper":"/paper/2506.02584","citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:0d9727412c34f2a7ef8eff77d20293d275fc073c9a5a56eaa0c9bb90b5a0baf5","observation_id":"4ccb6802-9dd2-4d7b-ab06-63f9b1e9196a","resolution":{"observed_at":"2026-08-07T11:26:02.787698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"cited_work":{"arxiv_id":"2506.02584","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02584","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prosodic struc- ture beyond lexical content: A study of self-supervised learning","venue":null,"work_id":"044c90e4-70d5-4951-bb6d-f93ca4fc2ae1","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-08-02T00:07:11.058562Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2506.02584","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:908a91166565414577c446131e2ced9919b16afd56340f4b4331572454864eea","observation_id":"d7cc8f92-7129-4540-b693-18d5efae5d6b","resolution":{"observed_at":"2026-05-10T13:35:26.588829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02584/citation-record","integrity":"/paper/2506.02584/integrity","json":"/paper/2506.02584/citation-record.json","paper":"/paper/2506.02584"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.421404Z","title":null,"venue":null,"work_id":"6e5083a8-06ad-47b2-a48c-f1d3c2a57226","year":null},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:02.555138Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:589ed1e7a4304524a415de0f47359f7344ee10caf572cbc4771bf6ed3d4fa9b9","observation_id":"3b99d18c-a814-4834-a2cb-77c676b926e7","resolution":{"observed_at":"2026-08-07T11:26:14.433966Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02584","snapshot_observed_at":"2026-08-07T11:26:02.787698Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:02.787698Z"},"links":{"cited_paper":"/paper/2506.02584","citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:0d9727412c34f2a7ef8eff77d20293d275fc073c9a5a56eaa0c9bb90b5a0baf5","observation_id":"4ccb6802-9dd2-4d7b-ab06-63f9b1e9196a","resolution":{"observed_at":"2026-08-07T11:26:02.787698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.346620Z","title":null,"venue":null,"work_id":"71b43eb8-4a59-43b0-b748-475865d29dab","year":null},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:02.942792Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:cfa2b0264b2f6f856e3e93a5d8bbe1d5fd4ef1ed6697525f2788cf0fd8aa9ed4","observation_id":"3aa3aa43-8c9a-43c8-a326-2b69fe162696","resolution":{"observed_at":"2026-08-07T11:26:14.350587Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.286889Z","title":null,"venue":null,"work_id":"82345de1-a3cf-4f76-86b8-35fd2dc4883f","year":null},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.173741Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:4ad47af84195ef286fa395a97f9ee55820811e6a112d0ca83018293fdefed366","observation_id":"b73be733-1810-45c2-bb8a-b0fc77023590","resolution":{"observed_at":"2026-08-07T11:26:14.309675Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.364692Z","title":"However, the extent to which this predictive capacity is contingent on the structure of lexical information or of its acoustic realisation is unclear","venue":null,"work_id":"0307c90a-0423-4ff9-b98b-9db0581ac1ce","year":null},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:02.638747Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:0d9416a38a2364e9c2012b552857fe7ee4084b5ff257cd01c29bd75829e604a7","observation_id":"a71c6dcc-c172-4776-9d07-c445507daff8","resolution":{"observed_at":"2026-08-07T11:26:14.390363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.881846Z","title":"How much does prosody help turn- taking? investigations using voice activity projection models,","venue":null,"work_id":"a9a8c985-aeaf-4109-80b1-bef699766270","year":2022},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.092897Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:ce48dbabc92113489710a62cd891a886f0353ef45e9f919a09839824f10416bb","observation_id":"9da63bd9-9b97-41ea-9e9d-069f82ee274c","resolution":{"observed_at":"2026-08-07T11:26:13.913481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.244418Z","title":null,"venue":null,"work_id":"481a87a9-87b5-473a-81e2-4279f288b93b","year":null},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.351671Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:e5ab26e1342110f13b499117deb27e2475618e278ee840e58ff9f0ce0d12e5cf","observation_id":"d6a4ded2-8bb2-4af4-a106-7e4a816510ec","resolution":{"observed_at":"2026-08-07T11:26:14.260928Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.205569Z","title":"A probabilistic Earley parser as a psycholinguistic model,","venue":null,"work_id":"be9788fd-8240-4044-8a4e-ac780374aa06","year":2001},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.467469Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:678745669bc4981505b38add21f115160d85d35b374595e8080cce4f368981d1","observation_id":"cea98d93-efb3-4cef-8820-60266b80403f","resolution":{"observed_at":"2026-08-07T11:26:14.219223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.150879Z","title":"Metrical expectations from preceding prosody influence percep- tion of lexical stress","venue":null,"work_id":"201ea2e9-6ce2-47a6-8598-fe2111d583fd","year":2015},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.625881Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:862eafaab2562483c685b39d573eb2b416424392e7ee1ff0fc047591d454a766","observation_id":"3a4e4fc0-ff3d-4841-8cb3-7def7905e973","resolution":{"observed_at":"2026-08-07T11:26:14.174678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.079643Z","title":"Quantifying the perceptual value of lexical and non-lexical channels in speech,","venue":null,"work_id":"ad7312a7-3c89-4c76-99db-3ccc20dcf626","year":2023},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.732043Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:c8ebbae4db7fdf9e3810e2b002594d5480772672b290a9a6761f1b9062a1264e","observation_id":"edbf8d0c-26be-4598-ab68-4d6650a7b9ec","resolution":{"observed_at":"2026-08-07T11:26:14.111790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.006557Z","title":"Intonation facilitates prediction of focus even in the presence of lexical tones,","venue":null,"work_id":"7ecbf2ce-a2c6-40c0-838e-cad79b41ec4e","year":2017},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.842756Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:03c63ac40672d8e0bbeba9a86e7b0379ce39f38b84aa468a430eeaa76c320aab","observation_id":"575ec6e3-ba1a-4490-8e9a-85454c112aa1","resolution":{"observed_at":"2026-08-07T11:26:14.039859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.946470Z","title":"How long is the sentence? Prediction and prosody in the online processing of language,","venue":null,"work_id":"ed6263ca-e49f-4bd6-b3aa-1ff9b5e23740","year":1983},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.952772Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:1d6a78d3383c5dd65f9650120679f44ee7bd245040e4c3cc484394253b6467a7","observation_id":"5a54bd08-d521-49db-93e5-34d02488e993","resolution":{"observed_at":"2026-08-07T11:26:13.966925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.597042Z","title":"The language of proteins: NLP, machine learning & protein sequences,","venue":null,"work_id":"282a11a2-b46b-414d-b1fc-5a4d1501be2f","year":2021},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.837593Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:d9eda1ece9e5a1e984a710feb454a49d24fe44433e06fcb5f8d5b9015c3de9e3","observation_id":"053bcceb-6e43-4e46-8dc4-4ab1a81c4c75","resolution":{"observed_at":"2026-08-07T11:26:13.605168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.836851Z","title":"How prosody is both mandatory and optional,","venue":null,"work_id":"f271a561-0be9-4112-8871-4ec2eaab8ace","year":2014},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.199321Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:1aed4557477850c726156519f4cfd596b585ad048ea1eef5bea23446541cbfda","observation_id":"b9d4fffc-dfd0-4d07-a6a6-183b5585035b","resolution":{"observed_at":"2026-08-07T11:26:13.869075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:14.318640Z","title":"For Wav2Vec and HuBERT, we use the model checkpoints trained on≈900h of LibriSpeech, which is comparable to the amount of data seen by the MPM","venue":null,"work_id":"1c4873cb-fa34-4795-a9e6-3383338bbf36","year":null},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:03.056981Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:af3758ad7aef2316d9c953bdd3fd58be43efdf02f2e1f841eeb3719b8b37e7ac","observation_id":"19ff2bbe-c59e-497a-8e58-7595c3dce6bb","resolution":{"observed_at":"2026-08-07T11:26:14.332806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.782663Z","title":"What do you mean, you’re uncertain? The interpretation of cue words and rising intonation in dialogue,","venue":null,"work_id":"0cd94c74-cd2d-4b15-870c-7a7da0d828da","year":2010},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.279230Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:a7e525681fe9d0300494943e13975d663a42e1a82cf3e71b28881f5cf2e0a022","observation_id":"bcd601b9-052e-4104-9fa3-deaa54faed6d","resolution":{"observed_at":"2026-08-07T11:26:13.796531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.748558Z","title":"Delexicalised auditory priming of implicit prosody,","venue":null,"work_id":"f59a1832-dc70-4704-b2a5-241561921618","year":2020},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.362781Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:848cc3e50107af0d85321173de779916d5ffc53840d246fa67ae1331dbc3a20a","observation_id":"7bed56d4-1048-4f96-89d5-c4e9b8f1f01d","resolution":{"observed_at":"2026-08-07T11:26:13.768956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.679508Z","title":"Towards automatic detection of reported speech in dialogue using prosodic cues","venue":null,"work_id":"cc2bb68a-5042-48fe-ba45-04adf124c9a4","year":2015},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.503933Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:ffef510a93155823063fa64b0acea958f3ebf1d894ca90b3103b46733a4c3628","observation_id":"4d07e10a-a529-4379-b2d6-685b41620d38","resolution":{"observed_at":"2026-08-07T11:26:13.709366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.629355Z","title":"DeCAF: A deep convolutional activation fea- ture for generic visual recognition,","venue":null,"work_id":"b4187ead-6147-47a7-8159-67ad472712b6","year":2014},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.591983Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:9c9a25500c5755ef5efc3cbc94bece60a6cd101086183640cf4f872d4c0a32a9","observation_id":"ff6bda2a-d35b-426a-87df-b670ba998646","resolution":{"observed_at":"2026-08-07T11:26:13.638818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.613212Z","title":"BERT: Pre- training of deep bidirectional transformers for language under- standing,","venue":null,"work_id":"d21c6442-61fa-4289-8d14-6a49a9e69acf","year":2019},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.730062Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:05d862eccb9db0c51a1785268aaf28cf64ff9aba77a266c78ade48aea82b6f8a","observation_id":"baaaa3a4-359d-41e0-adfa-8950552097e2","resolution":{"observed_at":"2026-08-07T11:26:13.621266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.05862","last_updated":"2019-09-11T08:19:49Z","snapshot_observed_at":"2026-08-06T05:33:37.746688Z","submitted_at":"2019-04-11T17:29:30Z","title":"wav2vec: Unsupervised Pre-training for Speech Recognition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.05862","snapshot_observed_at":"2026-08-07T11:26:04.946181Z","title":"wav2vec: Unsupervised pre-training for speech recognition,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:04.946181Z"},"links":{"cited_paper":"/paper/1904.05862","citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:7bad74ff4cc199bf2b02fcecff962942759232808a715bb9e70c435d2c5bcb37","observation_id":"bbba3595-c4d5-4e38-9e09-6c455e5d1628","resolution":{"observed_at":"2026-08-07T11:26:04.946181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.581894Z","title":"HuBERT: Self-supervised speech repre- sentation learning by masked prediction of hidden units,","venue":null,"work_id":"2a1d4995-c4f7-4c91-bcf6-ea957646729b","year":2021},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.025464Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:22c578baf4795b3298422dc28dafd6db0309e9c4456661afeb24ffa077f86dd0","observation_id":"1d5f8b1a-4aa5-49bc-abae-2ca608235ae7","resolution":{"observed_at":"2026-08-07T11:26:13.588019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.568244Z","title":"ProsAudit, a prosodic benchmark for self- supervised speech models,","venue":null,"work_id":"142d5523-324e-4712-b97b-3bfe40c72978","year":2023},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.129108Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:3db5d04a878e7b8d46ffa39fad404c2b9367fb277d6b36c0f85572d63ef3c8e6","observation_id":"765587e7-7e0c-4fe9-9442-79603e8d0959","resolution":{"observed_at":"2026-08-07T11:26:13.573954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.550726Z","title":"Towards end-to-end prosody transfer for expressive speech synthesis with Tacotron,","venue":null,"work_id":"fb247815-edd5-41aa-a18c-101225baed32","year":2018},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.237644Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:e7ed4296f81bfc9444203d608bc85e2013ebd18768454c33b82c5330edc90ac4","observation_id":"217867fc-54f4-49d9-80e9-601a6ad8f8a5","resolution":{"observed_at":"2026-08-07T11:26:13.558392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.533864Z","title":"Disentangling prosody representations with unsupervised speech reconstruction,","venue":null,"work_id":"53302213-49e0-4ebc-8546-a05e0c92dda7","year":2023},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.347157Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:a8ae83c534e58a38210af0e5d6c68cf8e1f59140f3b351f991cd095a126f7d7e","observation_id":"b88e6260-579c-4947-ab6c-290063870a4b","resolution":{"observed_at":"2026-08-07T11:26:13.542249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.517585Z","title":"Towards paralinguistic-only speech representations for end-to- end speech emotion recognition,","venue":null,"work_id":"cdbbea28-a598-4580-9d66-af1c3654d075","year":2023},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.457844Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:b55b749138218fc109aa9ecc2d70a9730ea42b9540c9057c128ce638558bded0","observation_id":"ccf2d19c-42a5-431c-8451-d065df18dc0b","resolution":{"observed_at":"2026-08-07T11:26:13.524480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.498159Z","title":"Do prosody transfer models transfer prosody?","venue":null,"work_id":"ea2bff00-9ef2-4fc3-97be-9ec3f8f00a8e","year":2023},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.547065Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:ed0dfccb94f888d45eb5109044f25111e7254887b99344bff5191422ceb6b535","observation_id":"3c9d0ce0-f977-4f81-a35f-50a91404ff3d","resolution":{"observed_at":"2026-08-07T11:26:13.507022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.480026Z","title":"Hierarchical rep- resentation and estimation of prosody using continuous wavelet transform,","venue":null,"work_id":"dbfa7a90-9573-4231-b7db-ded96f865bc8","year":2017},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.656185Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:07b14816563ec417deaf89fcdcff9f0520fa3312349205d360028dc9e2707b2c","observation_id":"19373b59-0e59-4624-a12a-39c3ed0f481e","resolution":{"observed_at":"2026-08-07T11:26:13.486422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.463114Z","title":"Prosodic promi- nence and boundaries in sequence-to-sequence speech synthesis,","venue":null,"work_id":"bf75489e-84f3-49aa-a902-dfa8718d6cd7","year":2020},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.795157Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:f0a491f389d3bb6afb282555b283a720db75643ce6be95f44692e51de4b8cd46","observation_id":"274e9bc7-0b26-4921-a58a-602d1a2056f8","resolution":{"observed_at":"2026-08-07T11:26:13.470824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.449022Z","title":null,"venue":null,"work_id":"d934befa-4056-458a-9f84-c44122cc48a9","year":1984},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:05.913060Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:484f28a8e6cf149c4bd4f6a9a9df635857a56ee0e1f78adc656064e3a6d32600","observation_id":"69a8d1c4-8037-464a-be0b-e72621b853e2","resolution":{"observed_at":"2026-08-07T11:26:13.454150Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.433647Z","title":"Beckman,Stress And Non-Stress Accent","venue":null,"work_id":"067f0163-9d3d-4a02-9825-153e011cfdb7","year":1986},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.014935Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:0b451df154573d77c10c932e42f767d49ed262d3c1d90fc927fe721c339adbcc","observation_id":"ae4e5dbb-d8de-46f2-9b0b-e9b07ab08708","resolution":{"observed_at":"2026-08-07T11:26:13.441309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.416153Z","title":"Automatic detection of sentence prominence in speech using predictability of word-level acoustic features,","venue":null,"work_id":"21429444-0165-42a1-8e74-42cfd2dcc0b4","year":2015},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.158116Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:91306049b739fcf35f65abb9e0bb79f07b8e1b2f0ea87a2da3f21c66397cbd83","observation_id":"c23f5831-9d67-4b96-8d30-dcb7b942e7fd","resolution":{"observed_at":"2026-08-07T11:26:13.423376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.398421Z","title":"Learn- ing de-identified representations of prosody from raw audio,","venue":null,"work_id":"5c0bece6-0723-4d28-b2f6-58895facd933","year":2021},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.236122Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:020949493154029f6997cc83799dd2fd33a34765f46cc27bdc16cb3890997922","observation_id":"664e6b8d-25d9-407f-b047-37abdeb97a7d","resolution":{"observed_at":"2026-08-07T11:26:13.406105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.382620Z","title":"PMI-Masking: Principled masking of correlated spans,","venue":null,"work_id":"f673680b-d32d-4ff2-a99b-28429984d4d0","year":2020},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.346744Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:c9c08393f2ee8b4e3ebd7488f5b55718cef9c9c00afa90de4796c8779d21c5ee","observation_id":"2aab41a6-3cad-4625-bbb7-a1450be564e1","resolution":{"observed_at":"2026-08-07T11:26:13.388983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.365989Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":"986ac03a-18b8-4be5-a166-4c343fe74651","year":2020},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.467845Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:03421424bb576c2663dec8d5ac42a4949fa018e9333bf2c3079ef8a28d8a6d93","observation_id":"ec0c20cd-b29d-41a1-b728-eef8421ac52c","resolution":{"observed_at":"2026-08-07T11:26:13.370182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.298522Z","title":"WORLD: a vocoder- based high-quality speech synthesis system for real-time applica- tions,","venue":null,"work_id":"38b86800-e770-470c-8a7c-4b8f57942312","year":2016},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.576243Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:89e58d286e0374b88fed0df4f29bd21c6cea166bb34e03298b20b3f17bc7781c","observation_id":"e7547dfa-b555-4980-80a0-c1a11fb65707","resolution":{"observed_at":"2026-08-07T11:26:13.352330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:13.120495Z","title":"Classification of prosodic events using quantized contour modeling,","venue":null,"work_id":"051d1cce-47d9-4875-904d-463ad161424b","year":2010},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.652871Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:15b0e52a30268d43f3a3e4060111ea38f402e1cb34342fac3465e5b8b76cf0be","observation_id":"4e2a1de1-5441-4b42-930c-058f7d78e377","resolution":{"observed_at":"2026-08-07T11:26:13.219269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:12.877329Z","title":"Pitch contour shape matters in mem- ory,","venue":null,"work_id":"901e81cf-3c45-4a80-8068-3e157e8dffe7","year":2016},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.792706Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:0e19ddf57a48ffef9c71490a9071a4452342361517f4191688f22c5aaf14a90b","observation_id":"5e370401-769b-44ee-ad0c-14c55ed86d3e","resolution":{"observed_at":"2026-08-07T11:26:12.993884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:06.902092Z","title":"How transferable are features in deep neural networks?","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.902092Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:78ffd8bde0c34233f1e1148527b9f5cf547e88bb967c6e603e639ff3418b2e77","observation_id":"b6e9688e-5d03-4ae0-b245-378040c9e144","resolution":{"observed_at":"2026-08-07T11:26:06.902092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:12.845726Z","title":"Understanding intermediate layers using linear classifier probes,","venue":null,"work_id":"bfd37c2d-05e5-4b33-a8f4-7ef70b02ef12","year":2017},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:06.980687Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:e5bd5a328f6ed2acd8e86e9e7178d966b7bb8f5f141d88a43c37e3e15464b2f8","observation_id":"c0ef8cbf-f524-4751-a2ea-7ba6c4b25414","resolution":{"observed_at":"2026-08-07T11:26:12.849077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:12.582195Z","title":null,"venue":null,"work_id":"6c2d3048-8e19-4c6f-a667-94a0161c40a2","year":2002},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.054682Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:6dbe925186b26ff3345488d9ae141f335d48e1b55380aea900d341f03f884efc","observation_id":"323ad57b-5c17-442f-a811-f60dc8aee503","resolution":{"observed_at":"2026-08-07T11:26:12.749605Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:12.246832Z","title":"Young infants’ retention of information about syllables,","venue":null,"work_id":"f53a2447-4688-4620-9874-65803beab6c0","year":1995},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.174209Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:4d93e4ad801590afa8879f2216399d84fc420ea55514d360a3fc3e3a1535c262","observation_id":"11a116b3-a9ed-4872-9979-a2dc48a7ee35","resolution":{"observed_at":"2026-08-07T11:26:12.409171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:11.839681Z","title":"The influence of different prosodic cues on word segmentation,","venue":null,"work_id":"6f55d1ec-a30d-4885-a280-c2bd1bba5093","year":2021},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.266210Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:245e9656d546c0d9ca5248998708df947787ac8d123b5ad9a4150ca246048da4","observation_id":"2e19569a-4356-40b3-96f3-431ad6d0842a","resolution":{"observed_at":"2026-08-07T11:26:12.070762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:11.525948Z","title":"Prosody in context: A review,","venue":null,"work_id":"73de9e06-9217-47b0-9a40-e9106c80feb3","year":2015},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.442133Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:decdd69c01dfe4ec3fd18bfd7b8e73c5f282e9a1c1a3447f500e96f4a38a4e03","observation_id":"70586d26-62a3-4849-a463-4524cbcf7249","resolution":{"observed_at":"2026-08-07T11:26:11.661807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:11.498167Z","title":"Segmenting into adequate units for automatic recognition of emotion-related episodes: a speech-based approach,","venue":null,"work_id":"bd54967f-d8d4-416f-8ccc-c131ef539d6a","year":2010},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.535101Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:31e3a2a9d1c84149925d120afa377bd70f48cdadb2a13e0e13060132af1c695b","observation_id":"dd0d634a-8d33-4e51-8645-98e7503ece87","resolution":{"observed_at":"2026-08-07T11:26:11.516127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:11.361515Z","title":"TIMIT acoustic phonetic continuous speech cor- pus,","venue":null,"work_id":"db4a0e83-5ce8-49c7-bc81-06e9d8f0d582","year":1993},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.612782Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:9f81b7a7da564036223997c1ef6c4cddb08b5dd823f87b766cd025df1f833e59","observation_id":"4bc69d55-3950-435b-839f-e9f729a9bd39","resolution":{"observed_at":"2026-08-07T11:26:11.421560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:11.212237Z","title":"Convex weighting criteria for speaking rate estimation,","venue":null,"work_id":"ff85e622-c316-4902-9ffc-9cc52657056b","year":2015},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.682869Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:87056ba9d292f537e26b0fcf3ce46785610d1207bd21e40fbe3fe56ee4a6c5bb","observation_id":"9e38ced0-a13e-424e-8dac-badc92a8dd37","resolution":{"observed_at":"2026-08-07T11:26:11.293583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.01122","last_updated":"2021-08-04T01:49:10Z","snapshot_observed_at":"2026-07-06T11:34:54.333802Z","submitted_at":"2021-08-02T18:47:59Z","title":"Automatic recognition of suprasegmentals in speech","version":2},"cited_work":{"arxiv_id":"2108.01122","doi":null,"metadata_source":"pith","pith_arxiv_id":"2108.01122","snapshot_observed_at":"2026-08-07T11:26:09.034232Z","title":"Automatic recognition of suprasegmentals in speech","venue":"cs.CL","work_id":"4581f86a-053c-4492-b24b-8c56a7d61030","year":2021},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.797102Z"},"links":{"cited_paper":"/paper/2108.01122","citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:de0e32b3cfe67c8ba1e1fcfd0c6253a4375a0dae8e28880d6f886e733d79b764","observation_id":"c0d6b2f1-92e7-4f0c-9e21-38ec091302cc","resolution":{"observed_at":"2026-08-07T11:26:09.095115Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:11.028022Z","title":"Pre-linguistic segmenta- tion of speech into syllable-like units,","venue":null,"work_id":"49c545c2-2820-4486-a667-c7beebb5fdcb","year":2018},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.864326Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:66a5a600b66be5ab27586945d6fa908b7919fd7dc06f05a84cbdf196b1755b4c","observation_id":"09f60b95-1233-418d-848e-500a560b0d99","resolution":{"observed_at":"2026-08-07T11:26:11.116585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:10.892195Z","title":"Contribution of prosody to the segmentation and storage of","venue":null,"work_id":"b59d0d47-a77e-4f7d-8cd8-c5556fdd5602","year":2002},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:07.943243Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:585eecad68b7eb0d34016500b0f97e17e4550ecc5bc46b9c48abd8f5d7663783","observation_id":"a081e4a0-a907-4da1-89b8-5f5b6e875ab1","resolution":{"observed_at":"2026-08-07T11:26:10.953851Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:10.728358Z","title":"The Boston University radio news corpus,","venue":null,"work_id":"a91eb5c1-ab30-43cc-a776-2479a373baf8","year":1995},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.015224Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:e46095b5d99cdea6f954295a4d6d54e48cf58525dbd892e6ccf4e044e05ca6ef","observation_id":"d1d3e721-4c63-4df7-bcf5-84baa67bcf7b","resolution":{"observed_at":"2026-08-07T11:26:10.818550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:10.587393Z","title":"ToBI: a standard for labeling English prosody,","venue":null,"work_id":"df9dbd3d-fbd1-4f22-9f83-4bcbc7a1af94","year":1992},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.148698Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:271dc60a91263f8fb79770c006bf30da1988ef233baf02c11ee84b2c26487da1","observation_id":"05de4b06-44b0-433d-8bb3-477b3e90d540","resolution":{"observed_at":"2026-08-07T11:26:10.652611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:10.422257Z","title":"Automatic prosodic event detection using acoustic, lexical, and syntactic evidence,","venue":null,"work_id":"16d1f654-992a-4a71-9818-ff413965aa7b","year":2007},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.225259Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:caa697dc81fa764564274804bc98b754930a8d040fc6c2eff79f685284dff95b","observation_id":"e335fa52-e7b9-4297-8f01-3d583235f5f5","resolution":{"observed_at":"2026-08-07T11:26:10.489694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:10.259673Z","title":"Towards low-resource prosodic boundary detection","venue":null,"work_id":"98f3f2a7-03da-4c67-a060-41b13875f077","year":2014},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.331442Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:b0d06cf418dac8f059e01374485e51b8e71ec95d13a7c4b23eb73c4ec475fe57","observation_id":"7b7cb352-f886-4996-8e89-b2d7ad7a966d","resolution":{"observed_at":"2026-08-07T11:26:10.324095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:10.078202Z","title":"A saliency-based auditory atten- tion model with applications to unsupervised prominent syllable detection in speech","venue":null,"work_id":"62cd387d-5595-4e26-8dc8-16cc44106cf8","year":2007},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.434500Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:769be41145956717e5fba1cebbd1646c01b9b5565a4c83c3ceac79831593a5c8","observation_id":"56d2b57a-1c47-4b45-b029-48d189a54a6b","resolution":{"observed_at":"2026-08-07T11:26:10.184340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.865072Z","title":"The Ryerson audio-visual database of emotional speech and song (RA VDESS): A dynamic, multimodal set of facial and vocal expressions in North American English,","venue":null,"work_id":"46593c38-1db0-4678-8060-f871b8b04374","year":2018},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.524879Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:bb6d5774fcb56444c98fa2d06ec2cab2bc6e56729af63b69a10d186e5d94d9a1","observation_id":"4adb3e52-2d39-4775-848f-2990a26c94df","resolution":{"observed_at":"2026-08-07T11:26:09.945616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.698630Z","title":"Prosodic cues for emotion: analysis with discrete characteriza- tion of intonation,","venue":null,"work_id":"d2915db4-a5f3-4e19-887b-dbd80d3adbb9","year":2014},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.590275Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:82608f0c11b26103c853f372db95893ee1a427fb54d67cf3794375be1cac78ea","observation_id":"ba4b09d4-efb1-4625-aa27-dff62ca3d217","resolution":{"observed_at":"2026-08-07T11:26:09.763452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.499996Z","title":"SpanBERT: Improving pre-training by representing and predicting spans,","venue":null,"work_id":"0f114467-f3cb-4150-b511-4edd47a044e5","year":2020},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.717814Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:8d6d6024c77a62db87e41b9258c047a9702c5b367b5f7bda76679685b1ff467a","observation_id":"98dd187c-c318-425f-b5d3-00409c9626ca","resolution":{"observed_at":"2026-08-07T11:26:09.608550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.352053Z","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech,","venue":null,"work_id":"162fa3f3-a403-4d31-89ed-8aa245771294","year":2019},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.792143Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:4372bcd9db60f9f5b9a8e8e0218c4179a25e435e99f1fc1a0d23465def129daf","observation_id":"f4b46df8-0fef-4e47-b9a5-c974b583d4b3","resolution":{"observed_at":"2026-08-07T11:26:09.425675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.188939Z","title":"Recognizing emotions in spoken dialogue with hierarchically fused acoustic and lexical features,","venue":null,"work_id":"fbea9859-3cab-4a08-94a4-082c04febef3","year":2016},"citing_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:08.869391Z"},"links":{"citing_paper":"/paper/2506.02584"},"observation_digest":"sha256:0b8e42349328599d19c750a0f30ec2e7bd5920dfdcef072e19f5b163f33166d4","observation_id":"75f74f29-ac55-4c99-93b6-ae22fdd7aa4e","resolution":{"observed_at":"2026-08-07T11:26:09.269131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T11:18:28.089348Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":50},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 2 inbound Pith citation observations for arXiv:2506.02584."}