{"as_of":"2026-08-08T08:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:379c020b4da050d005298ef114eadb7869a7f604e6163f5f50c872ead60ef1d0","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:40:11.848020Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:39:46.886300Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T11:40:12.056485Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"cited_work":{"arxiv_id":"2506.02088","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.02088","snapshot_observed_at":"2026-08-07T11:40:12.056485Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","venue":"cs.SD","work_id":"8001bf5c-13fe-4d95-890b-cb21d83bd18c","year":2025},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.886300Z"},"links":{"cited_paper":"/paper/2506.02088","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:5106d0307179bff122816f7c84926607c0d9162ab3a8b6a706d7089ee3c0a65c","observation_id":"b0fd2927-ab92-4ea7-b6d2-df576f3a353d","resolution":{"observed_at":"2026-08-07T11:40:12.074820Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02088/citation-record","integrity":"/paper/2506.02088/integrity","json":"/paper/2506.02088/citation-record.json","paper":"/paper/2506.02088"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.813882Z","title":"Early SER relied on hand-crafted features but struggled with real- world generalization [2]","venue":null,"work_id":"28a486ba-2220-47bd-8f91-a9a84090d304","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.828246Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:77031a057fa2c6f0ff5344fc4ff8e6065fc88a9cbe800baef5ff653cda6a7c1a","observation_id":"cbc91fd8-2696-431c-83e7-60c230cb87b5","resolution":{"observed_at":"2026-08-07T11:40:58.900210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"cited_work":{"arxiv_id":"2506.02088","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.02088","snapshot_observed_at":"2026-08-07T11:40:12.056485Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","venue":"cs.SD","work_id":"8001bf5c-13fe-4d95-890b-cb21d83bd18c","year":2025},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.886300Z"},"links":{"cited_paper":"/paper/2506.02088","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:5106d0307179bff122816f7c84926607c0d9162ab3a8b6a706d7089ee3c0a65c","observation_id":"b0fd2927-ab92-4ea7-b6d2-df576f3a353d","resolution":{"observed_at":"2026-08-07T11:40:12.074820Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.659330Z","title":"The hidden states of the last layerLof the text encoder are denoted byZ L T (j)for positions j= 1,","venue":null,"work_id":"91e78c0a-8895-4a59-93ca-a567c0d06b78","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.937316Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:b2feaab2ca4e45d1083fdea634725b8d254a968162b571cc8328653e4838ebde","observation_id":"d0e4715b-d7fe-41df-a432-fdb412559c61","resolution":{"observed_at":"2026-08-07T11:40:58.725410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.626319Z","title":null,"venue":null,"work_id":"a62edddf-e725-4da4-a364-7ec92f78a632","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.962061Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:bb4203747bb76cf0b9b31d3ccbfb9a1ac873687d434875faf73f2f1517116bd7","observation_id":"568c6db1-c9bf-445c-b9e1-92b773217c62","resolution":{"observed_at":"2026-08-07T11:40:58.649776Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.443731Z","title":"We report results for unimodal speech models, bi- modal fusion with text, prosodic and spectral feature integra- tion","venue":null,"work_id":"56593075-ba1b-4b70-8084-8883969b0890","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.019513Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:1a38363c7c4e2540022592457b92d0d2b7294f7a4e92a95cd81ffd18890b3eaa","observation_id":"33001ac9-da6f-4328-9c06-f8c518f3b824","resolution":{"observed_at":"2026-08-07T11:40:58.509032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.292421Z","title":"Our evaluation of unimodal models demonstrated the strong performance of Whisper and XEUS, highlighting their robustness for SER in spontaneous speech","venue":null,"work_id":"d47b58a4-ef1b-4a75-bccc-1eb08c1a166e","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.249227Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:fcfa29e88edb8295daa0b8a464d5cb0d3ecac329e9b159a8ab4c1104632ea6bd","observation_id":"a15c6334-fba5-4b40-977e-75b1be96e4a9","resolution":{"observed_at":"2026-08-07T11:40:58.356842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.153649Z","title":"We also thank the Artificial Intelligence Lab at Re- cod.ai, the Institute of Computing, University of Campinas","venue":null,"work_id":"022effbf-6002-4c36-a858-e2fe68378742","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.380433Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:99f696a5fc4f26062ea3787c49b3dfd1d6b0c6c4aeefb707b34696070733ed4f","observation_id":"8a356d0a-2d75-4065-b968-ab684529c6e7","resolution":{"observed_at":"2026-08-07T11:40:58.217464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.085467Z","title":"Affective computing mit press,","venue":null,"work_id":"e73d1712-5573-424a-a829-9cbb864bca08","year":1997},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.456692Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:6f099b2e36db27cad190c76e4072d84f9e07e565c86802677ff5d093b7e9e8d4","observation_id":"bd61b554-8a33-449d-b640-1d6fbccfcf00","resolution":{"observed_at":"2026-08-07T11:40:58.093220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:47.566911Z","title":"Iemocap: Interactive emotional dyadic motion capture database,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.566911Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:f81a170115f38ba99246dedfd18ccca18a5bdf66944f575ae7a1f9aadc4f24f1","observation_id":"95edbc09-aad6-4d1d-ae6e-d62d26767b08","resolution":{"observed_at":"2026-08-07T11:39:47.566911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.973677Z","title":"Every rating matters: Joint learning of subjective labels and individual annotators for speech emotion classification,","venue":null,"work_id":"bb3c803f-7f38-42bc-9c81-4c60c6bdaf7b","year":2019},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:48.879938Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:ea212f647b23c1f14fe0ec4b292791bae926b9d63957fdd9b98c59add014a0a1","observation_id":"575e1dbe-4310-41de-a8e4-6c02731c8573","resolution":{"observed_at":"2026-08-07T11:40:58.052371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:51.667321Z","title":"Speech emotion recognition using self-supervised features,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.667321Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:841cd1add6b1d57dca2729e8f5c5b308c593884d44ab20e8c2cf24d071f545fb","observation_id":"5a988f1c-8399-4005-9333-b47543d5203f","resolution":{"observed_at":"2026-08-07T11:39:51.667321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.900272Z","title":"Chakraborty, M","venue":null,"work_id":"fccf5fc0-a0f7-4055-b364-f80fe57a64a2","year":2017},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.789972Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:83772a46933a27ed6d1b6efb6627781d88ef4468769ad31dec6ce8d2e996da83","observation_id":"4aa89ce1-a259-4b5d-a722-c3b2640b180e","resolution":{"observed_at":"2026-08-07T11:40:57.908096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.796562Z","title":"Speech emotion recognition with multi-task learning,","venue":null,"work_id":"8cfcf4dc-7410-47b6-abd8-72d3b4ec4e71","year":2021},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.886757Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:49e09de4dc52cd2ada4705bbe058eef9b4cfdd1d65be06eeac0a7f5cee3a1f1e","observation_id":"f8ca3961-213f-4b2b-91ec-7514c4909d14","resolution":{"observed_at":"2026-08-07T11:40:57.829289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.708229Z","title":"Improving speech emotion recogni- tion using self-supervised learning with domain-specific audiovi- sual tasks,","venue":null,"work_id":"66c02912-f665-4b9a-8ab4-3c541e5bd203","year":2022},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.999237Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:265d238b590f443f425a3d46e72433cacb5c09dd9b11283cb6cc9afe189b0a90","observation_id":"af2b8935-da7c-4af9-ab15-bbac4f956498","resolution":{"observed_at":"2026-08-07T11:40:57.755073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:56.052863Z","title":"Odyssey 2024-speech emotion recognition challenge: Dataset, baseline framework, and results,","venue":null,"work_id":"3431ade5-553a-4f50-a76c-faec51528625","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.100807Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:337a9a506574c8b4051da0bbea7103c3aa0ab38514b83f70f5d3ea66633d305d","observation_id":"dc666eda-a686-4562-ac18-c5b4e6d3a2df","resolution":{"observed_at":"2026-08-07T11:40:57.518830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.179588Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech repre- sentations,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.179588Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:ccb1111f923eabfd4d01bf633c167b74b24e9fe6e08fb83851f8aac8b873cda4","observation_id":"e4a34ba8-918f-4d92-a487-dfd354d892a4","resolution":{"observed_at":"2026-08-07T11:39:52.179588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.259720Z","title":"Hubert: Self-supervised speech represen- tation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.259720Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:420de1deccbcc35d4ae47fbf929c5114601900fac58d2f300a2259a3e80c66c7","observation_id":"cafa7854-99b7-47db-b70b-92195a3418cd","resolution":{"observed_at":"2026-08-07T11:39:52.259720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.329379Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.329379Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:da7833dcc3cc2c21595faeef8424a8d4f294776467666a0974c50454436a4bc2","observation_id":"82513e37-8794-4a55-986b-5d009f0a68c3","resolution":{"observed_at":"2026-08-07T11:39:52.329379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.399609Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.399609Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:2bda44ded3ee0cc5f790ade20520d37e527615053a40b96b45310eafc470f51e","observation_id":"814768dd-203d-4df1-bc62-9d272e49614e","resolution":{"observed_at":"2026-08-07T11:39:52.399609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.465169Z","title":"Towards robust speech representation learning for thousands of languages,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.465169Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:8e3b46975638445402c1b5bc00a6da60b94086ae67c7d061ec5ffbba316a9a8f","observation_id":"609fc84d-5d93-4fe4-a39b-a0eb1dd6b082","resolution":{"observed_at":"2026-08-07T11:39:52.465169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.920125Z","title":"A robustly optimized BERT pre-training approach with post-training,","venue":null,"work_id":"5cc7a949-fb7e-4073-a3b5-903203b7bbe9","year":2021},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.554243Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:5aa1cb7dab9502759060abc5f678fc7bfe6d29027bb7a19165a0242bf2d2a64f","observation_id":"fb018c13-febc-43f5-9629-7294bac6d95a","resolution":{"observed_at":"2026-08-07T11:40:55.952387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01954","last_updated":"2023-05-03T08:11:25Z","snapshot_observed_at":"2026-08-07T01:52:48.386723Z","submitted_at":"2023-05-03T08:11:25Z","title":"SeqAug: Sequential Feature Resampling as a modality agnostic augmentation method","version":1},"cited_work":{"arxiv_id":"2305.01954","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.01954","snapshot_observed_at":"2026-08-07T11:40:12.003559Z","title":"SeqAug: Sequential Feature Resampling as a modality agnostic augmentation method","venue":"cs.CL","work_id":"6e99f6d9-251f-40ee-8455-41fd9343b685","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.777099Z"},"links":{"cited_paper":"/paper/2305.01954","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:a3daffb0a925fbaebc982597d8ab0fba67141f0ad9ee8c0fd67d1c04211470c6","observation_id":"aa5f7fc5-f77c-4c38-8d0c-800414f835c5","resolution":{"observed_at":"2026-08-07T11:40:12.025604Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:53.234553Z","title":"Enhancing cross-language multimodal emotion recognition with dual attention transform- ers,","venue":null,"work_id":"712c8579-29d9-4dd5-8b79-56c21511d91e","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.019255Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:f48e8fb9ead8616f578c647a1d0ccd0747c4b60842ffc831a1d60966154448d2","observation_id":"6c227393-77bb-4ef8-ad33-c932444ebe03","resolution":{"observed_at":"2026-08-07T11:40:54.883253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:36.106956Z","title":"Ced: Con- sistent ensemble distillation for audio tagging,","venue":null,"work_id":"26ac8abe-9879-4458-9a69-665c3a5bd411","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.103227Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:5c4f4e0073e2612d403dffbdf40b8488d1291f7405cc5b6da67ea740caf2a3aa","observation_id":"57f0aca7-3596-43b6-8b87-bf654988b716","resolution":{"observed_at":"2026-08-07T11:40:46.567879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06910","last_updated":"2024-01-09T11:45:34Z","snapshot_observed_at":"2026-07-06T15:15:34.051890Z","submitted_at":"2023-04-14T03:25:00Z","title":"HCAM -- Hierarchical Cross Attention Model for Multi-modal Emotion Recognition","version":2},"cited_work":{"arxiv_id":"2304.06910","doi":null,"metadata_source":"pith","pith_arxiv_id":"2304.06910","snapshot_observed_at":"2026-08-07T11:40:11.955002Z","title":"HCAM -- Hierarchical Cross Attention Model for Multi-modal Emotion Recognition","venue":"eess.AS","work_id":"d66b6061-c7da-48c1-83f1-a558fa78b36a","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.213116Z"},"links":{"cited_paper":"/paper/2304.06910","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:91f435767c8b33329f0ad09535939c3eca4ab35ea2cea4875646248f51a18eb0","observation_id":"cfa2a6ae-afd7-4b88-a60d-6e728af99904","resolution":{"observed_at":"2026-08-07T11:40:11.978878Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:35.993290Z","title":"Graph attention networks,","venue":null,"work_id":"81664a14-7051-4159-9066-495a895e2bcc","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.374984Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:bd1a74aa91d18dd320f61ed7f81e861d739401a618acf54771152647f85136ee","observation_id":"adf8d24e-0b38-43d3-8eb2-5c3958ba2a54","resolution":{"observed_at":"2026-08-07T11:40:36.044619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-07T11:40:07.102687Z","title":"Glu variants improve transformer,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:07.102687Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:59bf91a46356f69c07d9fae5f0eee51db1cd64af9c5d5c422442bfcfd81399ff","observation_id":"73217c7a-5c82-49d8-8a29-7fffc3b96f74","resolution":{"observed_at":"2026-08-07T11:40:07.102687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:35.812409Z","title":"Espnet: End-to-end speech pro- cessing toolkit,","venue":null,"work_id":"2d026bc1-3e4e-46b4-8a50-c5cca55acd4d","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:08.043740Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:1aa9fa9bb9d5fe12cc6cb5342c56e11cd86f31eef2c0363d41eedc06fbac4ea1","observation_id":"2e2d95a2-2ebd-46ec-8686-ddc1d47f9b94","resolution":{"observed_at":"2026-08-07T11:40:35.916452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:09.312368Z","title":"Less is more: Accu- rate speech recognition & translation without web-scale data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.312368Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:2e102b8ef231ff44c8a25ce3f5533d38d6c133341be9d64b2eceb4bff5fdf469","observation_id":"571a2128-c921-4234-92c9-21618fd4f1e6","resolution":{"observed_at":"2026-08-07T11:40:09.312368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.549542Z","title":"1st place solution to odyssey emotion recognition chal- lenge task1: Tackling class imbalance problem,","venue":null,"work_id":"ff5cfa22-8e4d-4ca5-9631-3a66fd76b149","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.642774Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:0d3a9f0666d91d75374d67c817fb17ab4fbba01256a454505f03774a8014189e","observation_id":"738e5e1b-7056-46d6-ab15-de24cbf70da5","resolution":{"observed_at":"2026-08-07T11:40:55.870025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:33.534336Z","title":"Fundamental frequency ex- traction in speech emotion recognition,","venue":null,"work_id":"d898edcb-437f-4db3-a406-07e453fdd95d","year":2012},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.779236Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:4e952d4c25d9fcf2db9e08a1e14048f5b0a0c0cb7053a49d01e23c88c85e403f","observation_id":"c2ef2ada-153e-4b7b-b0f4-8a58e4ec2f0f","resolution":{"observed_at":"2026-08-07T11:40:35.694011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:30.830689Z","title":"Autoregressive neural f0 model for statistical parametric speech synthesis,","venue":null,"work_id":"506b7420-7c15-4e05-b2a8-0047189c3e46","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.869009Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:7246c6cb89f75eaac45ef66b8410d5b96c810cbf6059d40c8ca0010f6402413c","observation_id":"b59c1520-ded0-4b57-84eb-320e3a08162b","resolution":{"observed_at":"2026-08-07T11:40:31.875345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:23.564544Z","title":"Rmvpe: A robust model for vocal pitch estimation in polyphonic music,","venue":null,"work_id":"1c9fff7c-6b3a-46bd-a442-b3ad964239f6","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.900326Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:98e679ae8f3f58e9d6b2c4b05f59ce5caff8377028affd5ec896da3d7454e41e","observation_id":"f9bb7dae-4b5b-4664-94b4-2ea372393358","resolution":{"observed_at":"2026-08-07T11:40:23.794768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:21.613155Z","title":"Enhancing skin can- cer diagnosis using swin transformer with hybrid shifted window- based multi-head self-attention and swiglu-based mlp,","venue":null,"work_id":"d77c6639-6288-45ea-a471-a6d3a30936b8","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.918778Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:12ce4031bd00168c6a051939ed63b639352d7f7f9aa9f6b2bbaab683312ce245","observation_id":"b5824415-23d1-4220-ac0e-c03afc0ea926","resolution":{"observed_at":"2026-08-07T11:40:23.219875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.214296Z","title":"Searching for activation functions,","venue":null,"work_id":"8782f5ca-43ca-412e-8c59-ff108c60ce46","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.980654Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:aea5d84e73a95ab655f369441c3df19c3e4f3b3bc646ccd0ab444d60e38fc294","observation_id":"61f5b1ec-e191-4293-afb4-86cceefebbc0","resolution":{"observed_at":"2026-08-07T11:40:12.252580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:11.029630Z","title":"Decoupled weight de- cay regularization,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.029630Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:d1de05b2a743231ecfd30f777d0f529f4fa19556c05c361440f6780cd0c62c1a","observation_id":"dec73175-a10b-4b58-8360-d52c414541d0","resolution":{"observed_at":"2026-08-07T11:40:11.029630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:11.634578Z","title":"Panns: Large-scale pretrained audio neural networks for audio pattern recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.634578Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:db7b23a258e22090f77974cf1634af70ed920dd63b6f0b306dec9d735533841e","observation_id":"072fc2c8-d40f-4cfd-8b25-d6b33dbf5e32","resolution":{"observed_at":"2026-08-07T11:40:11.634578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.111418Z","title":"Focal loss for dense object detection,","venue":null,"work_id":"68afa429-e4d8-4ac0-a60a-db1e52b2a1d8","year":2017},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.768796Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:f112ea7104ff35451e181e9d3de8558b52282d055ac2a8a24f733fa85d6d43dd","observation_id":"b66fe9de-cb41-4a86-b269-35a9afcb68dc","resolution":{"observed_at":"2026-08-07T11:40:12.140211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05672","last_updated":"2024-02-08T13:47:50Z","snapshot_observed_at":"2026-07-30T09:46:51.062184Z","submitted_at":"2024-02-08T13:47:50Z","title":"Multilingual E5 Text Embeddings: A Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05672","snapshot_observed_at":"2026-08-07T11:40:11.848020Z","title":"Multilingual e5 text embeddings: A technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.848020Z"},"links":{"cited_paper":"/paper/2402.05672","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:158603bc2cf7784901b0b42af6de87b264f9678a179fbc444186eb4e9f494bd4","observation_id":"75fdba96-0af3-4525-93be-145621894b8f","resolution":{"observed_at":"2026-08-07T11:40:11.848020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T11:33:39.566370Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":3,"verified_fuzzy":23},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 1 inbound Pith citation observation for arXiv:2506.02088."}