{"as_of":"2026-08-19T11:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ede92665a1dfdb3e14993341f13ab5fd9fdd51430d958ecf3c6fd474986f86bb","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T18:50:08.405900Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T18:50:08.261278Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-12T18:50:08.499324Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"cited_work":{"arxiv_id":"2411.11232","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.11232","snapshot_observed_at":"2026-08-12T18:50:08.499324Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","venue":"cs.SD","work_id":"8e46c1f2-bd7d-4d79-bf58-d19ed1dcf6cc","year":2024},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.261278Z"},"links":{"cited_paper":"/paper/2411.11232","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:ccf0c434b66743d3348f038d1dafd59924b565f6754712fb3a475a86c7d2837d","observation_id":"9029bd3b-988a-4b5d-a2ce-96aa0b219744","resolution":{"observed_at":"2026-08-12T18:50:08.503803Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.11232/citation-record","integrity":"/paper/2411.11232/integrity","json":"/paper/2411.11232/citation-record.json","paper":"/paper/2411.11232"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.902249Z","title":"Common ob- jective evaluation metrics, such as mel-cepstral distance (MCD)","venue":null,"work_id":"c982c004-fcc6-498d-98f2-508deee6ef91","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.253206Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:6186138cf02365236c57cc99ccaae36efd5c67f7501b2e0a49dfbde6c6f37fb3","observation_id":"a10d66e1-013c-4a02-a300-2a91e0e2666b","resolution":{"observed_at":"2026-08-12T18:50:08.906628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.888890Z","title":"As a result, some ob- jective measures or models related to human perception have been proposed [3, 4, 5, 6]","venue":null,"work_id":"7bedcb66-d445-4e8f-852a-27f0a5384a61","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.257531Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:e74dff3d126a04e987cd2719f90ac3df0a422e9daa212ff470995df59fa242f8","observation_id":"310222d8-cc72-43e2-8675-17cf92c5aab0","resolution":{"observed_at":"2026-08-12T18:50:08.893734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.853410Z","title":"Dataset In this paper, the experiments followed the same settings as the V oiceMOS Challenge 2022 [15]","venue":null,"work_id":"9224899e-e074-4127-a8d4-214976b97f5c","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.273361Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:46d0507ff5860a810ec256ae477af86a11ad39bf173a4041cc9603f5c0d68eb8","observation_id":"e87ed4fb-8977-4686-88e2-04c264b79dd2","resolution":{"observed_at":"2026-08-12T18:50:08.857226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.877118Z","title":"mean-listener","venue":null,"work_id":"6429d969-69dc-4bdd-b28d-71bd662c84ae","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.265344Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:2bf6f428f5144e3972db98d07fb1baaffbe6fbe1381ae2e0910d4641f1975eee","observation_id":"780f391c-0607-4a61-aa33-3478aefec9fb","resolution":{"observed_at":"2026-08-12T18:50:08.881238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.865387Z","title":"When the rater ID is not the mean-listener, the label representing the sample is the score given by the individual rater (an integer i from 1 to 5)","venue":null,"work_id":"57200f76-2173-4bed-b8ea-8e7c68383ab3","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.269386Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:f1835b754d9634ed96dc24710758f06895fc05fba73b2df477baf1a93ab525ad","observation_id":"ee4f1d43-f25c-4638-bcb1-238c065bba6e","resolution":{"observed_at":"2026-08-12T18:50:08.869423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.758506Z","title":"NISQA: A deep CNN-self-attention model for multidimensional speech quality prediction with crowdsourced datasets,","venue":null,"work_id":"d2dc65cf-80b4-4adc-8444-668c21a0f9db","year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.308516Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:4799f181c11dd35bc097ec61a381c10adfd00ea3297517b721aae8e978a1a2b4","observation_id":"1f5c6476-c622-4219-8e09-7e6423a05b1d","resolution":{"observed_at":"2026-08-12T18:50:08.762518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.842105Z","title":"Comparision with baseline methods We first compare the proposed SAMOS with the baselines","venue":null,"work_id":"95a6bcbe-f631-49bd-88fd-3522583f1ff0","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.277306Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:b09ab1b3321f0ef5c37262b6c34905971ffe1a307823f4b74328e41e11c978e8","observation_id":"67bc5e51-1fd6-407d-bb72-5fb44d2d6e40","resolution":{"observed_at":"2026-08-12T18:50:08.846329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.830352Z","title":"We can see that removing the semantic module resulted in the degradation of all the metrics on both datasets, indicating the importance of semantic repre- sentations from SSL model","venue":null,"work_id":"d0d41d79-7461-40ae-98b2-f7ec26a35fee","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.281046Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:9a28ff8c8069db4192f1e49089d66b92e770ea68a696ff3df6ee610a5d9d75f4","observation_id":"41c32c20-0186-47f1-a6b0-8cb3bec56c93","resolution":{"observed_at":"2026-08-12T18:50:08.834338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.817291Z","title":"To improve prediction accuracy, SAMOS employs parallel regression and classification heads, and finally outputs the final MOS score through an aggregation layer","venue":null,"work_id":"df92a532-ab10-4e0a-8454-9ec2c7dba3ef","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.285155Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:1ba0172c66935779d11302a87d7e6ddc419409255e9183a4e9793e3ee197e452","observation_id":"cdec2479-b405-4bf8-85b1-1d9413048786","resolution":{"observed_at":"2026-08-12T18:50:08.821712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.804285Z","title":"Mel-cepstral distance measure for objective speech quality assessment,","venue":null,"work_id":"fd22aadf-e09e-46df-924b-b7aeef0dbfcd","year":1993},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.288976Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:bfd6e1dcf3a8d444db91b60e2aa98ae248da47476ceed4a356c6cf22f1c597f5","observation_id":"a4a7314c-03f6-4e6b-9fbc-0c1eacc56cf3","resolution":{"observed_at":"2026-08-12T18:50:08.808890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"cited_work":{"arxiv_id":"2411.11232","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.11232","snapshot_observed_at":"2026-08-12T18:50:08.499324Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","venue":"cs.SD","work_id":"8e46c1f2-bd7d-4d79-bf58-d19ed1dcf6cc","year":2024},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.261278Z"},"links":{"cited_paper":"/paper/2411.11232","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:ccf0c434b66743d3348f038d1dafd59924b565f6754712fb3a475a86c7d2837d","observation_id":"9029bd3b-988a-4b5d-a2ce-96aa0b219744","resolution":{"observed_at":"2026-08-12T18:50:08.503803Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.292704Z","title":"SDR– half-baked or well done?","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.292704Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:0726b9d28ce3c61a16f47479acf31ba4ca36c7dae6582db4ca588d967704dbcf","observation_id":"30193a09-e986-462f-9cfc-2ca600386acb","resolution":{"observed_at":"2026-08-12T18:50:08.292704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.296480Z","title":"Per- ceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.296480Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:83ca279b23297a3676042ee7700d3c92a96ec90f6882fd8ac47b2faca44511ed","observation_id":"e07f6f7b-3792-4d31-b07a-1e0954f8979d","resolution":{"observed_at":"2026-08-12T18:50:08.296480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.300376Z","title":"A short- time objective intelligibility measure for time-frequency weighted noisy speech,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.300376Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:f5671d6d025c817f936466b128d3414a35971fd3e112084715f9abeb87e2ecc1","observation_id":"e44b9181-7668-4052-8766-de5ac55a6734","resolution":{"observed_at":"2026-08-12T18:50:08.300376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.770255Z","title":"ViSQOL: An objective speech quality model,","venue":null,"work_id":"7be79d1d-cfaa-4ba8-b87d-9ee747dc814b","year":2015},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.304873Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:42f1a168d1fb8929df1c584b92901c8eeb305ddf368f876e85d596e9ec786308","observation_id":"f594f578-5842-4e96-9dde-2ef897188c1c","resolution":{"observed_at":"2026-08-12T18:50:08.774239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1611.09207","last_updated":"2016-11-28T15:51:25Z","snapshot_observed_at":"2026-08-14T21:28:18.830898Z","submitted_at":"2016-11-28T15:51:25Z","title":"AutoMOS: Learning a non-intrusive assessor of naturalness-of-speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.09207","snapshot_observed_at":"2026-08-12T18:50:08.312306Z","title":"AutoMOS: Learning a non- intrusive assessor of naturalness-of-speech,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.312306Z"},"links":{"cited_paper":"/paper/1611.09207","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:f93e87bbad2a93fdd571513022da056a3308970a5167204e8ba366b3d2b17409","observation_id":"e699b29c-92a3-4aad-8d43-3aa357453f6f","resolution":{"observed_at":"2026-08-12T18:50:08.312306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.747475Z","title":"Quality-Net: An end-to-end non-intrusive speech quality assessment model based on blstm,","venue":null,"work_id":"a3f7a750-ff8d-4171-985d-c3288523c29e","year":2018},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.316397Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:cef15c407f0600077da0230fa9d6dc4955922c02ae9df76ba51040f33e5b1bb1","observation_id":"7b82dcd7-8716-4fbe-8279-c3b28f26b027","resolution":{"observed_at":"2026-08-12T18:50:08.751072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.320018Z","title":"MOSNet: Deep learning-based objec- tive assessment for voice conversion,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.320018Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:d0e7c77ce5702bad9f8f8a80beaff7a8053018bd17b701894360ebbbb2cd2435","observation_id":"3e78365f-4946-4647-898c-9c39a313e118","resolution":{"observed_at":"2026-08-12T18:50:08.320018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.323526Z","title":"MBNet: MOS prediction for synthesized speech with mean-bias network,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.323526Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:845aee4ead3c90951c4c6b890d9bbe9b10fb289823f9d220437260addab2142b","observation_id":"3d77dae0-f23a-4757-844d-9fa4cf8e0cbe","resolution":{"observed_at":"2026-08-12T18:50:08.323526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.720787Z","title":"LDNet: Unified listener dependent modeling in mos prediction for syn- thetic speech,","venue":null,"work_id":"44e8ebae-0be2-4ebd-a58c-766f2f202c7c","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.326959Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:7c2a45a1f0238fbf82e12bd65a11b8cb50eb48a40a3b9d7880f3aa727dd12963","observation_id":"794cebee-6b36-4bbb-b50d-89a86f18a8e7","resolution":{"observed_at":"2026-08-12T18:50:08.724766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.708305Z","title":"Generaliza- tion ability of mos prediction networks,","venue":null,"work_id":"c22b4a17-fe30-4c11-8f79-97ec27fbdbbf","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.330731Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:e342eadd8404a8ed26c88d5a57f2ec6ed04882a71d416c2ed58cbc1dfd1eb743","observation_id":"52e2e90f-d821-48e3-befb-f3f91db3124b","resolution":{"observed_at":"2026-08-12T18:50:08.712368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.695722Z","title":"Deep learning-based non-intrusive multi- objective speech assessment model with cross-domain features,","venue":null,"work_id":"a34e3bc5-2dbb-45e4-84cb-3b7edd27aa46","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.334617Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:899108171921523f607cb7b995824fc434e54e6c1287ef56eabb1bb0ab32f986","observation_id":"fb553b37-1660-4ac6-83a8-6a2253627619","resolution":{"observed_at":"2026-08-12T18:50:08.700005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.337864Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.337864Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:46ec923753e23cf0b15ae6729f0dbef8c1d049709a6f48ea296c97fd5f5bac2d","observation_id":"60af0c23-252a-4d64-97a0-0496ea529157","resolution":{"observed_at":"2026-08-12T18:50:08.337864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.675662Z","title":"The V oiceMOS Challenge 2022,","venue":null,"work_id":"2a4d9ac0-5818-4c94-ab03-282a76865c44","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.341380Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:752e3e6716ab26caa91a7a4ffbd504afaa394c5a0e5fba56a24a546678c0a71c","observation_id":"54fe6ef4-8be8-4958-8218-b5dc94710d9f","resolution":{"observed_at":"2026-08-12T18:50:08.679439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.664002Z","title":"A transfer and multi-task learning based approach for MOS predic- tion,","venue":null,"work_id":"14b26b01-8b32-4417-87ab-7d8a4eb3cde4","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.344950Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:d22fa86b59b779506a75f134a435719f638825f3c9c84c57839b73e38c5153d1","observation_id":"1c22bcfd-ac74-4282-8180-efc23733ff3e","resolution":{"observed_at":"2026-08-12T18:50:08.667934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.652258Z","title":"DDOS: A MOS predic- tion framework utilizing domain adaptive pre-training and distri- bution of opinion scores,","venue":null,"work_id":"ec444b00-9197-4517-95f9-5c5bb2ee2004","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.348461Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:d6abeca63303b06dcec54c80341f918bbe3a4caff41d33403dd47af84899f42a","observation_id":"25448d39-b0e7-4b40-8abc-220ca5382bb6","resolution":{"observed_at":"2026-08-12T18:50:08.656349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.639687Z","title":"UTMOS: UTokyo-SaruLab system for V oice- MOS Challenge 2022,","venue":null,"work_id":"fdd48745-64f2-492f-9b55-4bf80565ed40","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.351975Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:069302b80dfba7607e3c22aee953ce3902408d8ea710afd02d0c6a95ec6305c6","observation_id":"b4ec5239-740e-4ec7-912c-06563030db64","resolution":{"observed_at":"2026-08-12T18:50:08.643694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.627279Z","title":"Fusion of self-supervised learned models for MOS pre- diction,","venue":null,"work_id":"0dc7d1b9-351b-49dc-9fce-6ff1e61206e4","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.355250Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:a58c1aecafe95904ae03fe9f4f77ba3df732aa405bb6bf0bbb59fb9040345e68","observation_id":"c7776e9b-87ab-4c7b-9e29-10bdc810d981","resolution":{"observed_at":"2026-08-12T18:50:08.631520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.614236Z","title":"Ensem- ble of deep neural network models for MOS prediction,","venue":null,"work_id":"aaa9d78f-b3d0-4419-9909-ac8c58f3f217","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.359094Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:e2a30231d3a07d8f5defe46d88f698d40749579da5a570e631ee60c743fb6549","observation_id":"dded1169-beef-48c3-a257-46a1ba2d1be0","resolution":{"observed_at":"2026-08-12T18:50:08.618523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.601392Z","title":"RAMP: Retrieval- augmented MOS prediction via confidence-based dynamic weighting,","venue":null,"work_id":"d7809cef-e119-495f-9265-2f9b4da49f81","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.362859Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:4c024d5f0b4c5719bb1147a4d62096579872b7016bec4ff19a71c122e824a592","observation_id":"447a59fb-38f9-4672-a89c-98c72831c85d","resolution":{"observed_at":"2026-08-12T18:50:08.606030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.588867Z","title":"Investigating content-aware neural text-to-speech MOS prediction using prosodic and linguistic fea- tures,","venue":null,"work_id":"2d053273-e285-4f57-9c35-4dc69c6998cf","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.366284Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:860ee9197263f6f3010ea149e2b7d13a8ad1a7fd2a3fbe85c5a9057a74e49f7d","observation_id":"91352e5e-3eba-47d5-ab95-306699f19a0c","resolution":{"observed_at":"2026-08-12T18:50:08.593262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.576237Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech repre- sentations,","venue":null,"work_id":"6a123a3b-2b7e-43a8-af7a-70c0a818f055","year":2020},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.369808Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:989eb94f3bb3e652666fa53b0e7c9105d2fa5b1bdf0a1b5ff1f2968225367f71","observation_id":"1fee44e5-7b9f-45fb-8672-4163a1ba579f","resolution":{"observed_at":"2026-08-12T18:50:08.580675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02162","last_updated":"2024-06-04T09:51:02Z","snapshot_observed_at":"2026-08-17T20:05:01.194572Z","submitted_at":"2024-06-04T09:51:02Z","title":"BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation","version":1},"cited_work":{"arxiv_id":"2406.02162","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.02162","snapshot_observed_at":"2026-08-12T18:50:08.470377Z","title":"BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation","venue":"eess.AS","work_id":"43fc5d88-f339-4038-9ce1-49ba25d678d7","year":2024},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.373433Z"},"links":{"cited_paper":"/paper/2406.02162","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:de97385c6b134b88ba13f4597bff9b835ddb9156cce325978734a2c3ef521e27","observation_id":"fffca9b7-87f0-4c05-9ff7-43ce8e77bfb5","resolution":{"observed_at":"2026-08-12T18:50:08.475207Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.565389Z","title":"SQAT-LD: Speech quality assessment transformer utilizing listener depen- dent modeling for zero-shot out-of-domain MOS prediction,","venue":null,"work_id":"e1e9ada1-e384-4688-a88d-3e7ff186d421","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.377396Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:f969f1bb112c4f7106869ec83b7f596946810214032a5050e074194734c42a6e","observation_id":"1027c81d-02da-479f-ac32-dcac8a1dbf17","resolution":{"observed_at":"2026-08-12T18:50:08.569287Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.02373","last_updated":"2021-06-30T05:45:26Z","snapshot_observed_at":"2026-08-16T18:26:43.450918Z","submitted_at":"2021-05-05T23:53:27Z","title":"How do Voices from Past Speech Synthesis Challenges Compare Today?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.02373","snapshot_observed_at":"2026-08-12T18:50:08.381069Z","title":"How do voices from past speech synthesis challenges compare today?","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.381069Z"},"links":{"cited_paper":"/paper/2105.02373","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:c946e9f57bdcf8af05f5b99ef61c9f524e4a8a28e5653b62c9bb55fb5a3ebf5d","observation_id":"151e9e29-7922-43e7-982b-329479541b82","resolution":{"observed_at":"2026-08-12T18:50:08.381069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.385075Z","title":"The Blizzard Challenge 2019,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.385075Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:da1c6ebb0603a471952d79667592a4a2698bf8dc5409099202c69c710180cda5","observation_id":"02825d8f-0937-4a6c-8255-0eaafae92b5f","resolution":{"observed_at":"2026-08-12T18:50:08.385075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.389039Z","title":"Conformer: Convolution- augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.389039Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:a2020228671d1524c514e2f732244dbe195bebeebccc20b48bd8c0fc8d213d1c","observation_id":"3d4e407a-af34-485f-aa7f-04e59493193e","resolution":{"observed_at":"2026-08-12T18:50:08.389039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.539395Z","title":"ConvNeXt V2: Co-designing and scaling convnets with masked autoencoders,","venue":null,"work_id":"9af4c1be-61af-4e6e-86fd-a91700e11e6a","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.393014Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:1d719667ec465925a72eaecd2fcddacaab5b819be8a5d89a259ffbfe3658a805","observation_id":"0fd6d85c-a83c-4c24-98d7-879b107e970d","resolution":{"observed_at":"2026-08-12T18:50:08.543400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.11030","last_updated":"2022-04-23T09:19:16Z","snapshot_observed_at":"2026-08-18T08:27:26.160969Z","submitted_at":"2022-04-23T09:19:16Z","title":"Improving Self-Supervised Learning-based MOS Prediction Networks","version":1},"cited_work":{"arxiv_id":"2204.11030","doi":null,"metadata_source":"pith","pith_arxiv_id":"2204.11030","snapshot_observed_at":"2026-08-12T18:50:08.439406Z","title":"Improving Self-Supervised Learning-based MOS Prediction Networks","venue":"eess.AS","work_id":"9682a3fd-5e48-4926-84ef-d28692b53d9d","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.397560Z"},"links":{"cited_paper":"/paper/2204.11030","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:13711bf8eb1b5e62ddbec0c358204a6fc00cd6166e9dc153738453dff6cab48b","observation_id":"632c41a8-ad18-440c-8953-02c144402482","resolution":{"observed_at":"2026-08-12T18:50:08.445952Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.526722Z","title":"ESP- Net: End-to-end speech processing toolkit,","venue":null,"work_id":"47de4001-bbf6-480d-8ee5-5090ddd83556","year":2018},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.402117Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:fd228f90d6ee4206c732a730e39eda1be55716f94575ba02e169b6de42952250","observation_id":"567d4e74-2d84-44c6-a656-ca38ed976c65","resolution":{"observed_at":"2026-08-12T18:50:08.531013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.512667Z","title":"CSTR VCTK cor- pus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92),","venue":null,"work_id":"ebf35385-736c-402a-86f0-3f5995b41afd","year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.405900Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:6919ec6f78889884cc18908ff24fd66c4acee246d212732e268fd4e490fae161","observation_id":"b41c0c1c-fce8-4f73-bbbc-cc8666f5f234","resolution":{"observed_at":"2026-08-12T18:50:08.518311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-19T05:21:54.749369Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":3,"verified_fuzzy":28},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 1 inbound Pith citation observation for arXiv:2411.11232."}