{"as_of":"2026-08-07T20:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:02612d73c956e6ef2ae2ea1d0e8339727d10cb9ea56b60669b78145ced23bc0f","coverage":[{"denominator":81,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":81,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T05:03:22.360226Z","state":"measured"},{"denominator":81,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":81,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.02448/citation-record","integrity":"/paper/2508.02448/integrity","json":"/paper/2508.02448/citation-record.json","paper":"/paper/2508.02448"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:19.377166Z","title":"no agreement","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.377166Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:2928f47ce4dc5e7d517338315bc8830b54e2dabff114f61eb7b2061403857259","observation_id":"3c86e088-8876-425e-a24b-eb251ca703d4","resolution":{"observed_at":"2026-08-06T05:03:19.377166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.971868Z","title":"In many cases, OOD UAR is, surprisingly, higher than IID","venue":null,"work_id":"3db9729e-55b7-4c8a-b741-d529163d9f8c","year":null},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.383352Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:c7dddb5f96ca3cfd0326d806efa2f5c95ac5c8b87a048717c6068a70f98038d4","observation_id":"9de93712-50ae-4032-8861-3078be62e6f2","resolution":{"observed_at":"2026-08-06T05:03:38.038489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.829560Z","title":"We computed the centred kernel alignment (CKA) [53], a measure of similarity for hidden representations using EmoDB as a probing dataset due to its smaller size","venue":null,"work_id":"54ad4751-e975-489b-ac3c-7339f1aad03b","year":null},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.387975Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:0dc53d7be665b07fc3bf8a233ae0ef6e197480616bbf8f26edbc911d668db409","observation_id":"06e77628-a5a8-4cef-a7b4-6d0b9ddcc6fb","resolution":{"observed_at":"2026-08-06T05:03:37.886827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.696363Z","title":null,"venue":null,"work_id":"ad949c22-5e5a-47bb-a3ba-259f68e17afe","year":null},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.392856Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:79ebcc0b72f6917e8978d355127cc7da8937bcdf613f9542567f27fdfb36dab7","observation_id":"ffcd4b06-0b6b-4af2-a0d5-ad4bacc531c4","resolution":{"observed_at":"2026-08-06T05:03:37.744312Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.553500Z","title":"However, as before, the Spearman’s � between noisy UAR and year of publication (���), MACs ( ���), and � of parameters ( ���) was extremely low","venue":null,"work_id":"44f07982-494b-4d43-9071-9a39a22264b2","year":null},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.397688Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:c8888b608352a7d0e6dbb978c70b8d76460245dc1b85b351a7dec54a161d5c8e","observation_id":"6c3837b6-80f5-44ba-9f9d-2a94285ed2ea","resolution":{"observed_at":"2026-08-06T05:03:37.618366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.410925Z","title":"To do so, we computed the speaker-level performance for each task and used that as the utility to compute the Gini index","venue":null,"work_id":"fae7d69a-69c9-48ee-8401-0ba328c1bd11","year":null},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.402198Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:da031148cfe106253e6e3e76450a88a965f118d71a81f8958f02a360f4e28f5f","observation_id":"63894e64-f3dd-4212-9e7d-ddfa01cf5b16","resolution":{"observed_at":"2026-08-06T05:03:37.464255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.244514Z","title":"neural scaling laws","venue":null,"work_id":"08538b73-08b8-4222-ac6f-8308ceab387e","year":null},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.406943Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:e79f8a930572501e73026b1dbf9b909ed331607c40167cee2a31b8cbc18b87c0","observation_id":"2f346806-f418-4b9f-9434-be6c5ca93f94","resolution":{"observed_at":"2026-08-06T05:03:37.327739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:37.137456Z","title":"Speech emotion recognition: Two decades in a nutshell, benchmarks, and ongoing trends,","venue":null,"work_id":"bbce1a49-2589-498a-9a58-a75f9253c577","year":2018},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.412501Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:2bc0e84cb4ab0f4ac9fa150b94aa600bdc453d2e66ac783a58bcecf966a0d5e3","observation_id":"aede7709-3708-4475-83c5-8d09e98afae4","resolution":{"observed_at":"2026-08-06T05:03:37.188006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:19.417271Z","title":"Speech emotion recognition using deep learning techniques: A review,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.417271Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:25b1ace94032e3ea072be0eba585ed5654e70d4f957f72c73eca3a8ebe176fb0","observation_id":"a8d7c008-a5f9-4410-b002-4d057aea44c7","resolution":{"observed_at":"2026-08-06T05:03:19.417271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/odyssey.2024-35","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:22.611409Z","title":"Odyssey 2024 – speech emotion recognition challenge: Dataset, baseline framework, and results,","venue":null,"work_id":"969b7dd7-9238-4583-8a47-e106572f0560","year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.422484Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:129f5e638b84d575af567c4e6f17cb7cbfbf14a8256adc7a4b154bc4c6e5fd58","observation_id":"6c8e5dd0-4484-4431-b28f-dc2484369c5b","resolution":{"observed_at":"2026-08-06T05:03:22.737818Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:36.965732Z","title":"You BEEP Machine – Emotion in Automatic Speech Understanding Systems,","venue":null,"work_id":"e06b192c-67e3-4c24-9f5e-e1ddbd8314e2","year":1998},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.427084Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:be11a3096af45a460c42c9463241a6b9690a9994528e009d7aea3539453fd30f","observation_id":"e8f0fd03-a615-4a36-b5e3-f06fd89f7a25","resolution":{"observed_at":"2026-08-06T05:03:37.039782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:36.793338Z","title":"Emotion recognition in speech using neural networks,","venue":null,"work_id":"effd0099-a158-46a4-83d2-f9ed538b198c","year":2000},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.432637Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:342efd57c6d74c55030575e41dfab86be6347a41e6b15a80526785f77a0f3bf0","observation_id":"465cfdc3-ff28-4240-8513-8d2339404cee","resolution":{"observed_at":"2026-08-06T05:03:36.856424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:36.618586Z","title":"Adieu features? end-to-end speech emotion recognition using a deep convolutional recurrent network,","venue":null,"work_id":"eefbdf84-78cf-4e60-9645-a466089196e3","year":2016},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.437961Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:c5d056fa2797cf241a5960a2d9ac32209f09b59058882b4660c3bf23f689e843","observation_id":"ac44449e-5c6b-4f04-962e-12767bcac3ae","resolution":{"observed_at":"2026-08-06T05:03:36.700380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:36.479247Z","title":"Dawn of the transformer era in speech emotion recognition: Closing the valence gap,","venue":null,"work_id":"1d8f0497-ca8c-45db-94a0-addc7967bcd6","year":2023},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.442766Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a101eca28deda1eb7f27b0bc1cc511cfa571cdd874ed5121c847a0c7e2869371","observation_id":"d02c247f-63ea-401f-b5cd-96ef617c4cbb","resolution":{"observed_at":"2026-08-06T05:03:36.526978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:36.305522Z","title":"Hear: Holistic evaluation of audio representations,","venue":null,"work_id":"2200b601-d9c5-4063-91bb-636852225ba4","year":2022},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.447639Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:7db69d0fc4e91510591c315a4b391bd8b05c019f6d9607228d13a1eea9189965","observation_id":"b55ec4c2-03f1-4d7f-b1f4-ff495a1e9d9f","resolution":{"observed_at":"2026-08-06T05:03:36.402739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:19.452825Z","title":"SUPERB: Speech Processing Universal PERformance Benchmark,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.452825Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:fd804bea6497bcc052a12fa73a24148ac2ea33f61ef7f963f4b99236adfb3011","observation_id":"32c11d96-f5b5-4fe3-8766-fc3de223ee04","resolution":{"observed_at":"2026-08-06T05:03:19.452825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:36.125088Z","title":"Crema-d: Crowd-sourced emotional multimodal actors dataset,","venue":null,"work_id":"74b6f06a-de02-4fc0-b889-5d4560667812","year":2014},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.458352Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:5fe8bf71e9541d0c7ee55a964e1238b5b8cb6951b2d9aac6d06c71c1549103b0","observation_id":"2f1b721e-2847-414d-a15c-b179db631c95","resolution":{"observed_at":"2026-08-06T05:03:36.194559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:35.943150Z","title":"Iemocap: Interactive emotional dyadic motion capture database,","venue":null,"work_id":"e6b31da1-e0de-4e37-b8cd-6dd5484649e3","year":2008},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.463490Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:241d50e7e5d5171c2be7418a0950e42151c2f5d015a143a382803d15468643db","observation_id":"535373da-64ac-48b6-b9e1-31621122754f","resolution":{"observed_at":"2026-08-06T05:03:36.030006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:35.771484Z","title":"Probing speech emotion recognition transformers for linguistic knowledge,","venue":null,"work_id":"58d03085-e1fb-4aab-811b-8e40538ed984","year":2022},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.469588Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a8d408d9bad5f6e12af0ee4813d309826622fe2f46a663c0af5d422ee95b8434","observation_id":"5571f54d-bf58-4553-adde-98ab3192b8a2","resolution":{"observed_at":"2026-08-06T05:03:35.863708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:35.610008Z","title":"Interspeech 2009 emotion challenge revisited: Benchmarking 15 years of progress in speech emotion recognition,","venue":null,"work_id":"7f050323-e595-40dc-97a6-17625c2df3b3","year":2009},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.476425Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:4c3ea5322739a828ce329bb6a3c5bc155a751b97dfc7b6a5c03281e27d6f466f","observation_id":"2322f7a5-1c59-4614-adfa-1738ca49a58c","resolution":{"observed_at":"2026-08-06T05:03:35.657867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:23.713520Z","title":"Building naturalistic emotionally balanced speech corpus by retrieving emotional speech from existing podcast recordings,","venue":null,"work_id":"614e47cd-b944-4fbc-b3f7-b7342dd899bb","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.481776Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a875bd25335e734d430d1babf81269d439decd26d948fa09a7d1e2215928897c","observation_id":"9d0c612f-7023-4693-8c0a-7c06d81acb2f","resolution":{"observed_at":"2026-08-06T05:03:23.946796Z","resolver_source":"arxiv_id_nonexistent","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:35.437133Z","title":"A database of german emotional speech,","venue":null,"work_id":"2458cf79-d1f8-4968-b72a-4084ab8ea694","year":2005},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.486305Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:e668ec023bf8d12c707b2fb98423b95933f6e96cfa4cc92bb224c6c0510a43d5","observation_id":"9e6a1ad0-389e-4836-9a3d-9e21c473b455","resolution":{"observed_at":"2026-08-06T05:03:35.530815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:35.280360Z","title":"Releasing a thoroughly annotated and processed spontaneous emotional database: The fau aibo emotion corpus,","venue":null,"work_id":"0543401f-8dbb-4a55-8064-d0b964e19996","year":2008},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.490775Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:03d1281ee2f9f830321132cb211841c71c2b882c53daf0073e9543a378a9bbc0","observation_id":"3f5515ba-fd52-49ce-8be9-81dc2de75a9d","resolution":{"observed_at":"2026-08-06T05:03:35.339490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:35.046911Z","title":"The Interspeech 2009 Emotion Challenge,","venue":null,"work_id":"bace69c3-c4b8-4156-aa57-909c8856ac15","year":2009},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.495279Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:c6e4413650af189e1eb608362baaff3b700386e4fc86b202c2e508decd4efa67","observation_id":"6d422279-cd83-4f7b-9d95-a3848f4ab81c","resolution":{"observed_at":"2026-08-06T05:03:35.171599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:34.907790Z","title":"Sewa db: A rich database for audio-visual emotion and sentiment research in the wild,","venue":null,"work_id":"a60bdc23-a592-4ec6-a4bd-77822b0c746e","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.500194Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:861f855a35f9b0d79f0eaeb63e09962b3b2d4b9f15df360fb54d077e30381af6","observation_id":"02ac205f-98dc-4b9c-b82b-142035ad9e2d","resolution":{"observed_at":"2026-08-06T05:03:34.963714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:34.713857Z","title":"The ryerson audio-visual database of emotional speech and song (ravdess): A dynamic, multimodal set of facial and vocal expressions in north american english,","venue":null,"work_id":"98bea241-6b92-4994-b3a0-2b6317e53a45","year":2018},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.505340Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:2fac0e0205dd71744ba0aba45c47f2c1c1ad0e1e80f3362ea8a261f7430b9c9b","observation_id":"e28eefe0-6884-4d40-8326-60a215afd330","resolution":{"observed_at":"2026-08-06T05:03:34.789421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:34.523461Z","title":"Emotion recognition using a hierarchical binary decision tree approach,","venue":null,"work_id":"a1a60ce6-7a53-452a-9c97-86a5e6950f40","year":2011},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.511972Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:9ae988a466dc1441b0a7bd073518a202895b86dcc0e79e4a6829240789ea5053","observation_id":"ee951382-6880-4f91-8f91-d8baeb889ddf","resolution":{"observed_at":"2026-08-06T05:03:34.618726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:34.309043Z","title":"The bitter lesson,","venue":null,"work_id":"9ec6192d-c8a3-4397-a5ca-a969d875b411","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.517711Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:0f35ac1b22f815c65aee448b6786f095062fdc70bacaef5adddbd304965a8ab8","observation_id":"1f05f92a-9e99-4c42-9fb5-0d9cdaffb199","resolution":{"observed_at":"2026-08-06T05:03:34.418326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:34.109219Z","title":"The Geneva minimalistic acoustic parameter set (GeMAPS) for voice research and affective computing,","venue":null,"work_id":"8825b837-e2d8-4b82-b2ae-326e3a630b6e","year":2015},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.523490Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:9990200fb18553e2b8bb09f1885649745b011ea2c69c9ef9faec2b319891377c","observation_id":"58a90056-4e8a-46a4-b80e-d0f38f0ba0ce","resolution":{"observed_at":"2026-08-06T05:03:34.170561Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:33.934130Z","title":"An Image-based Deep Spectrum Feature Representation for the Recognition of Emotional Speech,","venue":null,"work_id":"0b9625dc-492b-4a24-b4d7-96e2dd2bf8b4","year":2017},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.529768Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:ab632e6489c2762066f1f74b9586be73210c76e4773831329ef99d46c08c6bdb","observation_id":"9a111804-0ed9-45e7-a496-e46af95ff60e","resolution":{"observed_at":"2026-08-06T05:03:34.038394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:33.591068Z","title":"Exploring deep spectrum representations via attention- based recurrent and convolutional neural networks for speech emotion recognition,","venue":null,"work_id":"773f036a-d2d9-4bcd-bdbe-38462b571d68","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.535562Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:14f8a7ed114001e5beddf8600c97d07e6fc63550e209b935d81bc94b8e5fa83e","observation_id":"0ac77706-1550-43e7-8d53-17ae967dc763","resolution":{"observed_at":"2026-08-06T05:03:33.781219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:33.358979Z","title":"AST: Audio Spectrogram Transformer,","venue":null,"work_id":"edc6ea5d-ac79-4211-bd33-92d4a1985718","year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.540383Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:8d7db15093af0936117f0eb2d536b3d8e3f82ca2caa267ddb7145477f4cf00f3","observation_id":"8923ae90-2738-4cd1-9be0-85c57d0e2962","resolution":{"observed_at":"2026-08-06T05:03:33.475809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:33.091477Z","title":"Audio set: An ontology and human-labeled dataset for audio events,","venue":null,"work_id":"573ff430-ffc6-450f-9e8b-9118ffe88b44","year":2017},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.546884Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:4c55aa4e7de2f3800afb239469467a0240170ebf50453082fbd5baf4f2a28257","observation_id":"e931c805-abaf-4292-85a1-266d28534600","resolution":{"observed_at":"2026-08-06T05:03:33.243327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:32.750885Z","title":"ECAPA-TDNN: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification,","venue":null,"work_id":"74a4eae2-36b1-496b-9eea-69918c8fff4d","year":2020},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.552857Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:ef284585e6f63e258a2c998c5bbc35ed9d65a2eee3eef8b9f5a5ca8d46d4e3f6","observation_id":"165fef91-0fca-46aa-a9dd-3c3f455f747d","resolution":{"observed_at":"2026-08-06T05:03:32.894033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:32.409935Z","title":"Panns: Large-scale pretrained audio neural networks for audio pattern recognition,","venue":null,"work_id":"126353f5-1ed3-4c30-9d20-81dc7eccc14e","year":2020},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.559189Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:11ad6679444be9f5970e18ba41a6d54b512ea21110185077d4ce031bb857ccd2","observation_id":"99bf963c-da43-4abb-8b76-9c213c88ef29","resolution":{"observed_at":"2026-08-06T05:03:32.513494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:32.037109Z","title":"The role of task and acoustic similarity in audio transfer learning: Insights from the speech emotion recognition case,","venue":null,"work_id":"f54c6a97-e7fa-4700-aa4b-d95dc601a85b","year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.565002Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:5eee18f4eb075d2ee498281535f397dca99f9557286e02771228cdcee407e4b6","observation_id":"1b441607-afe4-4027-9015-17fb85dc1e6d","resolution":{"observed_at":"2026-08-06T05:03:32.191822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:31.753174Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"61cd65bc-c669-4472-b715-995938773354","year":2023},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.570227Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:b695ca8bbbf14b4e00abe0b3062ccbc850cfc2316b107ef413f41b34e0e2233d","observation_id":"fe352456-1474-40f6-bb40-c7d366a8696c","resolution":{"observed_at":"2026-08-06T05:03:31.847246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:31.417358Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":"26cfb54e-bcf3-432f-865a-7ed14e22374d","year":2020},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.576878Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a4f39141f6b034e9a40f08fbb830059a687f4df45730affab0066e3571a1753a","observation_id":"50189dd4-8096-4312-b5db-33ad23075429","resolution":{"observed_at":"2026-08-06T05:03:31.519036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:19.582137Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.582137Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:b46638ce0e4e7a87f1ed2a88f12a6ed65a40adf949a7c9cf24d4a5f40a97e203","observation_id":"8d479d57-4a63-4a7f-a74f-e04e7652f9a7","resolution":{"observed_at":"2026-08-06T05:03:19.582137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:31.103607Z","title":"“You stupid tin box","venue":null,"work_id":"2892b189-a728-42ee-a00b-10ca10bc2a55","year":2004},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.587331Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:aa6a1fdd71741f046e5acced03a501c5be8cba2ba197ad79b07b7c1c9b90df42","observation_id":"b19a71b7-1c46-4286-b224-34dc632d1f1b","resolution":{"observed_at":"2026-08-06T05:03:31.287251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:30.802445Z","title":"Steidl, Automatic Classification of Emotion-Related User States in Spontaneous Children’s Speech","venue":null,"work_id":"3c7251c2-b1fa-470d-8d38-5fe8a049105a","year":2009},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.594185Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:caa11be8c5e8782f51a42deb27794cbd23c917f87e9cb3b1e17f12860343fa90","observation_id":"5949d547-d7d5-4eb6-a090-28466f48831d","resolution":{"observed_at":"2026-08-06T05:03:30.947264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.11943","last_updated":"2025-04-10T13:51:44Z","snapshot_observed_at":"2026-07-06T20:07:54.727859Z","submitted_at":"2024-12-16T16:25:58Z","title":"autrainer: A Modular and Extensible Deep Learning Toolkit for Computer Audition Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.11943","snapshot_observed_at":"2026-08-06T05:03:19.599518Z","title":"Autrainer: A modular and extensible deep learning toolkit for computer audition tasks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.599518Z"},"links":{"cited_paper":"/paper/2412.11943","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:adb279c2df7fd5d69969544537bfe3485ca604971941f76c7d6468ca2611146d","observation_id":"336e50d2-ea65-4e82-9918-aa43e14e1f2a","resolution":{"observed_at":"2026-08-06T05:03:19.599518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:30.469459Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"33fe9667-ccde-4022-8a5c-750a5948d6fa","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.605588Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:06120f806afea03be68165ecb00c0d0a6a3a1c734ab9e7cf679352a962cdbcbd","observation_id":"d3b0dda4-77bd-4931-a7e6-7663e1577982","resolution":{"observed_at":"2026-08-06T05:03:30.646043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.11692","last_updated":"2019-07-26T17:48:29Z","snapshot_observed_at":"2026-07-31T22:31:37.910868Z","submitted_at":"2019-07-26T17:48:29Z","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.11692","snapshot_observed_at":"2026-08-06T05:03:19.610885Z","title":"Roberta: A robustly optimized bert pretraining approach,","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.610885Z"},"links":{"cited_paper":"/paper/1907.11692","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:01d4b7c25253c884f5b4f0d7f7f4e2b29a37754224d34673a932ee72b5d6e8d0","observation_id":"47d2668d-17b1-4e17-8845-67af78d60956","resolution":{"observed_at":"2026-08-06T05:03:19.610885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:30.189549Z","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter,","venue":null,"work_id":"38c8b15f-d9a4-4797-aa7b-568fc9e12290","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.616889Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:e1dead612562d5edfeb09fef964a4746a430134ddc96def93632b61498ba9835","observation_id":"660c4b51-eff2-4d5c-ad98-eb33788ae010","resolution":{"observed_at":"2026-08-06T05:03:30.311960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:29.970149Z","title":"Electra: Pre- training text encoders as discriminators rather than generators,","venue":null,"work_id":"4a92e5be-8cf4-4367-986c-81ee4e08c79d","year":2020},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.621416Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:fb47060cc5e84ba79c52b003de797612e5676fed4b7099ff7d90ce4050ee1714","observation_id":"c83ecf1b-aafb-4d69-89dd-9c42b2611aa2","resolution":{"observed_at":"2026-08-06T05:03:30.063045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T05:03:19.627042Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.627042Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a1c5e99da0b12bb638d6511db8cbe90b38c56ad8bfd1c6b0f9aa5967335a3ff8","observation_id":"21584cc0-dafd-4e53-822f-e239d36b7728","resolution":{"observed_at":"2026-08-06T05:03:19.627042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T05:03:19.636024Z","title":"The Llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.636024Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:f10e08335b1d914303d20425d7224358e1bde0ac4296c82e9e265d6aefef816f","observation_id":"2d4fcd32-74b8-4368-bd8e-3d83a58013dd","resolution":{"observed_at":"2026-08-06T05:03:19.636024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-06T05:03:19.646510Z","title":"Mistral 7b,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.646510Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:58b45f212b02f945b38e51246d0f64a9045def5d0b915aca87ff4455d949371a","observation_id":"de4aa256-9ad1-41bb-8353-11cbe5877b20","resolution":{"observed_at":"2026-08-06T05:03:19.646510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:29.708350Z","title":"LoRA: Low-rank adaptation of large language models,","venue":null,"work_id":"f046c580-a909-4038-814f-a0272aaa932a","year":2022},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.657522Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:0de3ea316dac860214ed1c1e123a0f4d662c3268596b5d50412f08ba4cd70064","observation_id":"bd718021-0a75-469d-a0f7-a7ca82eaaa92","resolution":{"observed_at":"2026-08-06T05:03:29.820184Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:29.416612Z","title":"A curated dataset of urban scenes for audio-visual scene analysis,","venue":null,"work_id":"61d45bbb-1998-486b-b835-dc29f53699e3","year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.677420Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:6ce32e22f9f020d882ec47fb99773b5c90c136f614056d589f041ae428ea2394","observation_id":"e52ef06c-fed7-4052-8a3f-e26ddb41bfd6","resolution":{"observed_at":"2026-08-06T05:03:29.579282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:29.209503Z","title":"Enrolment-based person- alisation for improving individual-level fairness in speech emotion recognition,","venue":null,"work_id":"a81972db-4816-4a5e-9605-089d664ed115","year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.691640Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a489bd27ec22b14c1899d40a89348a280cad6c1a76fb994d33d1d7eb16895daa","observation_id":"9724ba06-5131-45cd-9576-a93777d65767","resolution":{"observed_at":"2026-08-06T05:03:29.303436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:29.020055Z","title":"What size test set gives good error rate estimates?","venue":null,"work_id":"5299a83c-55c2-462b-962f-8ff3ed9120e1","year":1998},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.702540Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:0449422f5a15f62ddce4a417b3842bdd9408bfc6f8825d373cd625a740d97c00","observation_id":"ee28d161-60b9-46bc-92fb-c1c71894eb40","resolution":{"observed_at":"2026-08-06T05:03:29.115752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:28.808642Z","title":"A formula for the gini coefficient,","venue":null,"work_id":"2dca3743-ffaf-4d2c-8ae8-4c64214433a0","year":1979},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.717678Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:23317db1d55bca43afe14c7403bbc662b934ada320e4adb9e20873584aad61a6","observation_id":"08875a01-7667-4041-9f7e-d81f92c9299d","resolution":{"observed_at":"2026-08-06T05:03:28.911427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:28.632061Z","title":"Shalev-Shwartz and S","venue":null,"work_id":"f240cf36-9b81-4728-bf85-88cdcacb4e5d","year":2014},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.731173Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:21567ce4dad3f976c0b28bd4eae8b3c334c232661e5df170f2bf5b9d010b4403","observation_id":"82549b11-8441-4ded-8e48-d7e133ea4819","resolution":{"observed_at":"2026-08-06T05:03:28.721923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:28.443839Z","title":"Underspecification presents challenges for credibility in modern machine learning,","venue":null,"work_id":"c90ed1f6-8992-4871-ae6f-a1dd57d1cc7d","year":2022},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.744077Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:0e58f57d7aa214363af8b970cf081c85c81f5c97e9cc941c67ca40d36f8c507f","observation_id":"81ebef54-8ee1-4d6e-80e8-4f7207303c8e","resolution":{"observed_at":"2026-08-06T05:03:28.515732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:28.229563Z","title":"On the power of curriculum learning in training deep networks,","venue":null,"work_id":"5f233e82-db68-468a-a21b-78832007f4cc","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.781937Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:32645f1de09bbbc92cdbd0320a32b7485fc29e34b990ba8a8df52b3b07ff4e17","observation_id":"43a78f07-3daa-4788-a861-5fd2d52e6ad0","resolution":{"observed_at":"2026-08-06T05:03:28.335926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00973","last_updated":"2024-11-01T18:55:31Z","snapshot_observed_at":"2026-07-06T19:43:50.724897Z","submitted_at":"2024-11-01T18:55:31Z","title":"Does the Definition of Difficulty Matter? Scoring Functions and their Role for Curriculum Learning","version":1},"cited_work":{"arxiv_id":"2411.00973","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.00973","snapshot_observed_at":"2026-08-06T05:03:23.105262Z","title":"Does the Definition of Difficulty Matter? Scoring Functions and their Role for Curriculum Learning","venue":"cs.LG","work_id":"3982f158-2744-41d5-801e-4f9c709cb3e6","year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.880732Z"},"links":{"cited_paper":"/paper/2411.00973","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:5b0b8a96da59e42b98ead9e958686706487ab02cc4e6749e9d0cbd77d0354884","observation_id":"a5f2549b-9788-4870-89a9-8b59e0638e4b","resolution":{"observed_at":"2026-08-06T05:03:23.242678Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:27.976321Z","title":"Scalable hyperparameter transfer learning,","venue":null,"work_id":"02a9e05a-0983-4daa-82c3-8b15914e0a51","year":2018},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:19.979770Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:4eeb7482337fdf268a9e0ba5a7bd738d2fad38aa4d84b96e1e2da314cbdcb313","observation_id":"dae10e57-27ab-4095-b50e-5094d5cb4d3f","resolution":{"observed_at":"2026-08-06T05:03:28.084329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:27.746719Z","title":"Similarity of neural network representations revisited,","venue":null,"work_id":"2e6adf00-5c64-4ee8-a8d7-59b843ca474d","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.054791Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:21f917c454ddd4f7b28c5a7ab048a29ad3dcfde867c625cfdfd1ea44daf80bf8","observation_id":"26caac61-68fc-4eac-baab-c2351cbafb40","resolution":{"observed_at":"2026-08-06T05:03:27.823419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:27.545575Z","title":"Deep learning of representations for unsupervised and transfer learning,","venue":null,"work_id":"b1ddd76e-12a6-4873-9996-12fb44f08710","year":2012},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.184801Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:930c127c5828910e7a9b97c7453484bda21474c9ac9bebbec0f04c8bc67410ac","observation_id":"9e6c16e1-e0e5-42a2-96f1-8ae7db0202a5","resolution":{"observed_at":"2026-08-06T05:03:27.614547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.11154","last_updated":"2020-11-13T19:09:09Z","snapshot_observed_at":"2026-08-07T10:59:37.724632Z","submitted_at":"2020-07-22T01:31:44Z","title":"Rethinking CNN Models for Audio Classification","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.11154","snapshot_observed_at":"2026-08-06T05:03:20.280689Z","title":"Rethinking cnn models for audio classification,","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.280689Z"},"links":{"cited_paper":"/paper/2007.11154","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:6bc0b4a5609790e3e6f6a124003fa5651e2d9cfa155cd5b681a266b54582e489","observation_id":"b90c8b98-a3a0-43cf-86ed-8e479b2a4bb1","resolution":{"observed_at":"2026-08-06T05:03:20.280689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:27.352479Z","title":"What is being transferred in transfer learning?","venue":null,"work_id":"806ccd18-8d20-45ee-bf11-94e4a2775f90","year":2020},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.381035Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:3cbc6c081e574033192fc981ca10fb3fe66b07b15a9acf8a013f4804304758d7","observation_id":"0312dece-7b8f-4f6e-bf13-553b1f9bcdb5","resolution":{"observed_at":"2026-08-06T05:03:27.463993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:27.141958Z","title":"Acoustic profiles in vocal emotion expression.,","venue":null,"work_id":"fb4f5107-9759-45c4-8ead-5faeef79fb80","year":1996},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.500209Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:2d2b7a4ab7155f1299fd5ff711bdc1a79a904ead5d8c19aa8a474f969a981f99","observation_id":"ee1d35ff-0990-47e3-9d03-5597d71d88a2","resolution":{"observed_at":"2026-08-06T05:03:27.242890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-06T05:03:20.587528Z","title":"Scaling laws for neural language models,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.587528Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:24e32b2766c7983d9bf1ba56b2f05bd2a9cf696fe6fdfc17e003defd673962f9","observation_id":"8cdbf087-1872-46aa-a7d4-65cc6bd9f841","resolution":{"observed_at":"2026-08-06T05:03:20.587528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15672","last_updated":"2025-07-28T13:58:00Z","snapshot_observed_at":"2026-07-06T18:50:04.824067Z","submitted_at":"2024-07-22T14:41:29Z","title":"Computer Audition: From Task-Specific Machine Learning to Foundation Models","version":2},"cited_work":{"arxiv_id":"2407.15672","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.15672","snapshot_observed_at":"2026-08-06T05:03:22.877147Z","title":"Computer Audition: From Task-Specific Machine Learning to Foundation Models","venue":"cs.SD","work_id":"8b75dfde-048b-4632-942e-9508bfc3187e","year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.694853Z"},"links":{"cited_paper":"/paper/2407.15672","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:3c9de6db170744f3eec7bdc8405affaa66bb5038f078f9eb18cb4a79029b8cc4","observation_id":"557500d8-a3f1-4f94-bc00-04d03185c0f0","resolution":{"observed_at":"2026-08-06T05:03:22.962325Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:26.929799Z","title":"Can large language models aid in annotating speech emotional data? uncovering new frontiers [research frontier],","venue":null,"work_id":"fda26fbc-9653-4d27-b057-b12b082d5c52","year":2025},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.837511Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:004a0bffb34a6199cd566967473978b903f152f72acabba57024a51c5f245e19","observation_id":"9a5d045f-0711-43bc-9225-a0d43aaf3647","resolution":{"observed_at":"2026-08-06T05:03:27.041394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:26.720204Z","title":"Winner’s curse? on pace, progress, and empirical rigor,","venue":null,"work_id":"8b59ad0e-d8d1-4d6b-b122-b028f1bf6125","year":2018},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:20.938232Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:ad0d69de97a1a45c01ae2bba7e3dc22e91b0e0766b0ed05f033dc760daacdced","observation_id":"6aa1f68f-f1a7-4eb3-9eb9-0e8cfb812a0b","resolution":{"observed_at":"2026-08-06T05:03:26.809672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:26.593263Z","title":"Troubling trends in machine learning scholarship: Some ml papers suffer from flaws that could mislead the public and stymie future research.,","venue":null,"work_id":"01b0f0b2-4580-48a1-95c3-15f995a678bd","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.025558Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:b9938e40b72d8b7083dac9a1e99bfed79b96b51c95a62ba07e85e0129d8a8a83","observation_id":"a7902f98-eb6d-4281-8907-7a4ee874fd61","resolution":{"observed_at":"2026-08-06T05:03:26.651612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.05446","last_updated":"2020-06-16T00:58:12Z","snapshot_observed_at":"2026-08-06T08:58:02.295940Z","submitted_at":"2019-10-11T23:51:09Z","title":"On Empirical Comparisons of Optimizers for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.05446","snapshot_observed_at":"2026-08-06T05:03:21.143206Z","title":"On empirical comparisons of optimizers for deep learning,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.143206Z"},"links":{"cited_paper":"/paper/1910.05446","citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:ff7970654b12c03bb2d7e02b9c58abab53c84a57ae3c7233e97679bff8759758","observation_id":"48f6f2a4-3fb4-499a-9cf4-0f33040fa121","resolution":{"observed_at":"2026-08-06T05:03:21.143206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:26.378217Z","title":"Unreproducible research is reproducible,","venue":null,"work_id":"765e8102-24e7-47cf-bf3b-85a638089136","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.218486Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:19fb61d5d55f778f26b7c1c8c619b0b8e08c0ad9c1175dc7e78773927e92ff88","observation_id":"6e497e63-4069-49b6-a63b-7dd12ec71bdf","resolution":{"observed_at":"2026-08-06T05:03:26.481829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:26.190560Z","title":"Beyond deep learning: Charting the next frontiers of affective computing,","venue":null,"work_id":"fd8cdd08-f23d-41d5-ad8c-ebe12d5884d6","year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.370321Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:3a84d2664a8d7bf2d9f5e3389730a9e08fbc11daf6f6169dbe992a9a0d27c6cf","observation_id":"44ba4015-41eb-4ba4-b92c-df3c81ed36a0","resolution":{"observed_at":"2026-08-06T05:03:26.272093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:26.011024Z","title":"Basic emotions,","venue":null,"work_id":"d730ba64-de36-4a7c-bbb0-61d6f60fe3c8","year":1999},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.502683Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:bc8f5ed6b23f4470fe24d752dd549d27e25f24bf6643d26bea135d33730391d6","observation_id":"8c2fd22b-2e59-4996-9d1e-0d968eab38fb","resolution":{"observed_at":"2026-08-06T05:03:26.072177Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:25.840934Z","title":null,"venue":null,"work_id":"2a089d9d-7e63-4b28-bf72-13c5131e1d58","year":2017},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.608418Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a702ef1c539597255efc96e6075d7ae164d379310f74b42eebd942d185b34be0","observation_id":"002e10c2-1783-4e9e-9c95-c3ed3fd7ad99","resolution":{"observed_at":"2026-08-06T05:03:25.885779Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:25.656867Z","title":"End-to-end speech emotion recognition using deep neural networks,","venue":null,"work_id":"0a57f9d5-f759-4124-baf9-0f4dd7ce0302","year":2018},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.756624Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a82c95e2644e47574b640064dc38555c9d69f9ee43ab2d5180a665c90a072a48","observation_id":"8f3741fc-9170-4329-b9cc-73542e2be759","resolution":{"observed_at":"2026-08-06T05:03:25.724323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:25.478090Z","title":"Speech emotion recognition using deep 1d & 2d cnn lstm networks,","venue":null,"work_id":"84a43cc4-8c1d-448c-b434-d5e6061ac6a5","year":2019},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.849418Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:b8d38ddb5da028177126835a61505a494b4748235103ff99361b46c327d8601b","observation_id":"2ca17b07-c16b-4d8a-a2b9-d70213058828","resolution":{"observed_at":"2026-08-06T05:03:25.558121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:25.222963Z","title":"V oxpopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation,","venue":null,"work_id":"4bb1a47b-f703-46d6-81ec-3012a672f2f5","year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:21.961338Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:a95b3cac42edac380bcbcd98ae5d2c3beccd8babfee62e41483286b6d38b1a80","observation_id":"75a9af9f-8537-410b-809c-63da0040fbc1","resolution":{"observed_at":"2026-08-06T05:03:25.367123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:25.045413Z","title":"Robust wav2vec 2.0: Analyzing Domain Shift in Self-Supervised Pre-Training,","venue":null,"work_id":"f6145c4f-05fb-457e-b2a8-13ec1eb02a75","year":2021},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:22.110325Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:8dfeb4a9f0db0b6fabdc2e5fbb066c2bbcb76143947d918fa60662bc32a5e25c","observation_id":"09d67d0f-c6d4-4047-8aa7-96cef23409fd","resolution":{"observed_at":"2026-08-06T05:03:25.127563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:24.725320Z","title":"Mp3 and aac explained,","venue":null,"work_id":"18f36798-446e-4940-ab84-284a2a66f5c0","year":1999},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:22.183466Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:cfbae6dfe4c32d43ed89f156298fc94f2f83913279bc90d992fb8d51e8474939","observation_id":"366485df-b7de-4888-afa4-c6dd4b258146","resolution":{"observed_at":"2026-08-06T05:03:24.894328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:24.409031Z","title":"High fidelity neural audio compression,","venue":null,"work_id":"9d5c325f-f77b-4a21-84b5-09f5cf20530d","year":2022},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:22.254389Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:1b54d2020ce240faebeb84c2240cca1c1318dd4ec498631b22c09cc576340fe5","observation_id":"5111cf98-f387-40f5-8be5-29d5bd67e7d4","resolution":{"observed_at":"2026-08-06T05:03:24.511956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:03:24.109810Z","title":"Semanticodec: An ultra low bitrate semantic audio codec for general sound,","venue":null,"work_id":"0ad8c28e-553e-4b65-b296-b4dbe1ce8dfd","year":2024},"citing_paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T05:03:22.360226Z"},"links":{"citing_paper":"/paper/2508.02448"},"observation_digest":"sha256:7e0dfa7f51341459387970c41dc9a489bd239d938dc54f68c9c6f228d26728fa","observation_id":"f9dce89f-c40a-4e17-bb14-135636930f05","resolution":{"observed_at":"2026-08-06T05:03:24.280485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.02448","last_updated":"2025-08-04T14:09:53Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T05:03:05.531743Z","submitted_at":"2025-08-04T14:09:53Z","title":"Charting 15 years of progress in deep learning for speech emotion recognition: A replication study"},"reference_resolution":{"displayed":81,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":4,"verified_fuzzy":63},"total_outbound_references":81},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 81 of 81 outbound references and 0 inbound Pith citation observations for arXiv:2508.02448."}