{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GBA25JCPQ5T4NC4ZJAF43HMHVW","short_pith_number":"pith:GBA25JCP","schema_version":"1.0","canonical_sha256":"3041aea44f8767c68b99480bcd9d87ad8ce4f2f21934cc6f5830cb50ee510b55","source":{"kind":"arxiv","id":"2505.14648","version":1},"attestation_state":"computed","paper":{"title":"Vox-Profile: A Speech Foundation Model Benchmark for Characterizing Diverse Speaker and Speech Traits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Anfeng Xu, Dani Byrd, Helin Wang, Jihwan Lee, Laureano Moro-Velazquez, Najim Dehak, Shrikanth Narayanan, Thanathai Lertpetchpun, Thomas Thebaud, Tiantian Feng, Xuan Shi, Yoonjeong Lee","submitted_at":"2025-05-20T17:36:41Z","abstract_excerpt":"We introduce Vox-Profile, a comprehensive benchmark to characterize rich speaker and speech traits using speech foundation models. Unlike existing works that focus on a single dimension of speaker traits, Vox-Profile provides holistic and multi-dimensional profiles that reflect both static speaker traits (e.g., age, sex, accent) and dynamic speech properties (e.g., emotion, speech flow). This benchmark is grounded in speech science and linguistics, developed with domain experts to accurately index speaker and speech characteristics. We report benchmark experiments using over 15 publicly availa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14648","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-05-20T17:36:41Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"4df7c189b4dd781d05895b4916b6dcbb49fff48c2114acfd248be1ed33687140","abstract_canon_sha256":"1c77daea4b0bbcbf0077a790a2ff7c878b69bd8409b75dfa56f0ec7eb3dc7f41"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:17.085511Z","signature_b64":"Bfus68Ta19/dVANkuFVMYqE/rSOVyzh4F3WKg9Y1Tj0tT1EWQ8te9jZteGwAzkeRHxcYk0PMq01XMDP7V/wfCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3041aea44f8767c68b99480bcd9d87ad8ce4f2f21934cc6f5830cb50ee510b55","last_reissued_at":"2026-07-05T11:06:17.084933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:17.084933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vox-Profile: A Speech Foundation Model Benchmark for Characterizing Diverse Speaker and Speech Traits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Anfeng Xu, Dani Byrd, Helin Wang, Jihwan Lee, Laureano Moro-Velazquez, Najim Dehak, Shrikanth Narayanan, Thanathai Lertpetchpun, Thomas Thebaud, Tiantian Feng, Xuan Shi, Yoonjeong Lee","submitted_at":"2025-05-20T17:36:41Z","abstract_excerpt":"We introduce Vox-Profile, a comprehensive benchmark to characterize rich speaker and speech traits using speech foundation models. Unlike existing works that focus on a single dimension of speaker traits, Vox-Profile provides holistic and multi-dimensional profiles that reflect both static speaker traits (e.g., age, sex, accent) and dynamic speech properties (e.g., emotion, speech flow). This benchmark is grounded in speech science and linguistics, developed with domain experts to accurately index speaker and speech characteristics. We report benchmark experiments using over 15 publicly availa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14648","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14648/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14648","created_at":"2026-07-05T11:06:17.084994+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14648v1","created_at":"2026-07-05T11:06:17.084994+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14648","created_at":"2026-07-05T11:06:17.084994+00:00"},{"alias_kind":"pith_short_12","alias_value":"GBA25JCPQ5T4","created_at":"2026-07-05T11:06:17.084994+00:00"},{"alias_kind":"pith_short_16","alias_value":"GBA25JCPQ5T4NC4Z","created_at":"2026-07-05T11:06:17.084994+00:00"},{"alias_kind":"pith_short_8","alias_value":"GBA25JCP","created_at":"2026-07-05T11:06:17.084994+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05365","citing_title":"SPEARBench: A Benchmark for Naturalness Evaluation in Streaming Speech-to-Speech Language Models","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2607.02214","citing_title":"Unlocking Speech-Text Compositional Powers: Instruction-Following Speech Language Models without Instruction Tuning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03283","citing_title":"SpeakerCard-1M: An Evidence-Grounded Corpus for In-the-Wild Speaker Verification","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31729","citing_title":"Is Natural Always Appropriate? Investigating Naturalness and Appropriateness Across Different Domains for TTS Evaluation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31055","citing_title":"Reference-Based Prosody and Rhythm Evaluation for Spoken Dialogue Systems","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03283","citing_title":"SpeakerCard-1M: An Evidence-Grounded Corpus for In-the-Wild Speaker Verification","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30543","citing_title":"TRACE: Temporal Relationship-Aware Conversational Entrainment Detection in Dyadic Speech","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05201","citing_title":"Exploring Speech Foundation Models for Speaker Diarization Across Lifespan","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23578","citing_title":"Style Amnesia: Investigating Speaking Style Degradation and Mitigation in Multi-Turn Spoken Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12036","citing_title":"Towards Fine-Grained Multi-Dimensional Speech Understanding: Data Pipeline, Benchmark, and Model","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19019","citing_title":"Smiling Regulates Emotion During Traumatic Recollection","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08363","citing_title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05201","citing_title":"Exploring Speech Foundation Models for Speaker Diarization Across Lifespan","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW","json":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW.json","graph_json":"https://pith.science/api/pith-number/GBA25JCPQ5T4NC4ZJAF43HMHVW/graph.json","events_json":"https://pith.science/api/pith-number/GBA25JCPQ5T4NC4ZJAF43HMHVW/events.json","paper":"https://pith.science/paper/GBA25JCP"},"agent_actions":{"view_html":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW","download_json":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW.json","view_paper":"https://pith.science/paper/GBA25JCP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14648&json=true","fetch_graph":"https://pith.science/api/pith-number/GBA25JCPQ5T4NC4ZJAF43HMHVW/graph.json","fetch_events":"https://pith.science/api/pith-number/GBA25JCPQ5T4NC4ZJAF43HMHVW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW/action/storage_attestation","attest_author":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW/action/author_attestation","sign_citation":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW/action/citation_signature","submit_replication":"https://pith.science/pith/GBA25JCPQ5T4NC4ZJAF43HMHVW/action/replication_record"}},"created_at":"2026-07-05T11:06:17.084994+00:00","updated_at":"2026-07-05T11:06:17.084994+00:00"}