{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DWNSGIXP5XYQKPRW6LQHFIOU55","short_pith_number":"pith:DWNSGIXP","schema_version":"1.0","canonical_sha256":"1d9b2322efedf1053e36f2e072a1d4ef7add418f5c78c832e85f27c5bd3c74f4","source":{"kind":"arxiv","id":"2504.20447","version":1},"attestation_state":"computed","paper":{"title":"APG-MOS: Auditory Perception Guided-MOS Predictor for Synthetic Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Hua Huang, Lizhi Wang, Zhicheng Lian","submitted_at":"2025-04-29T05:45:09Z","abstract_excerpt":"Automatic speech quality assessment aims to quantify subjective human perception of speech through computational models to reduce the need for labor-consuming manual evaluations. While models based on deep learning have achieved progress in predicting mean opinion scores (MOS) to assess synthetic speech, the neglect of fundamental auditory perception mechanisms limits consistency with human judgments. To address this issue, we propose an auditory perception guided-MOS prediction model (APG-MOS) that synergistically integrates auditory modeling with semantic analysis to enhance consistency with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20447","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-04-29T05:45:09Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"e314f8cef8b88502978673bda80f96328b88f4842f248482a45bfd649b787dcd","abstract_canon_sha256":"beb2bf3b8e7049e4fc2afd07cc66b8c392b58df62e2cdf4fdd1cbe2bda56c73f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:42.935337Z","signature_b64":"/A7sVPhMLdqveyQasmuOAmoTyvIRSEBErGc1g47b8Cl+8cdaTn2MI0+UvHzmfHZQoVBnvJHBqmxZEYx8EGrsDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d9b2322efedf1053e36f2e072a1d4ef7add418f5c78c832e85f27c5bd3c74f4","last_reissued_at":"2026-07-05T10:55:42.934803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:42.934803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"APG-MOS: Auditory Perception Guided-MOS Predictor for Synthetic Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Hua Huang, Lizhi Wang, Zhicheng Lian","submitted_at":"2025-04-29T05:45:09Z","abstract_excerpt":"Automatic speech quality assessment aims to quantify subjective human perception of speech through computational models to reduce the need for labor-consuming manual evaluations. While models based on deep learning have achieved progress in predicting mean opinion scores (MOS) to assess synthetic speech, the neglect of fundamental auditory perception mechanisms limits consistency with human judgments. To address this issue, we propose an auditory perception guided-MOS prediction model (APG-MOS) that synergistically integrates auditory modeling with semantic analysis to enhance consistency with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20447","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20447/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20447","created_at":"2026-07-05T10:55:42.934874+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20447v1","created_at":"2026-07-05T10:55:42.934874+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20447","created_at":"2026-07-05T10:55:42.934874+00:00"},{"alias_kind":"pith_short_12","alias_value":"DWNSGIXP5XYQ","created_at":"2026-07-05T10:55:42.934874+00:00"},{"alias_kind":"pith_short_16","alias_value":"DWNSGIXP5XYQKPRW","created_at":"2026-07-05T10:55:42.934874+00:00"},{"alias_kind":"pith_short_8","alias_value":"DWNSGIXP","created_at":"2026-07-05T10:55:42.934874+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00400","citing_title":"Deep Learning for Personalized Binaural Audio Reproduction","ref_index":130,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55","json":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55.json","graph_json":"https://pith.science/api/pith-number/DWNSGIXP5XYQKPRW6LQHFIOU55/graph.json","events_json":"https://pith.science/api/pith-number/DWNSGIXP5XYQKPRW6LQHFIOU55/events.json","paper":"https://pith.science/paper/DWNSGIXP"},"agent_actions":{"view_html":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55","download_json":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55.json","view_paper":"https://pith.science/paper/DWNSGIXP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20447&json=true","fetch_graph":"https://pith.science/api/pith-number/DWNSGIXP5XYQKPRW6LQHFIOU55/graph.json","fetch_events":"https://pith.science/api/pith-number/DWNSGIXP5XYQKPRW6LQHFIOU55/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55/action/storage_attestation","attest_author":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55/action/author_attestation","sign_citation":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55/action/citation_signature","submit_replication":"https://pith.science/pith/DWNSGIXP5XYQKPRW6LQHFIOU55/action/replication_record"}},"created_at":"2026-07-05T10:55:42.934874+00:00","updated_at":"2026-07-05T10:55:42.934874+00:00"}