{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:XLP2AMWVYL2ILRFO265JLGLHGL","short_pith_number":"pith:XLP2AMWV","schema_version":"1.0","canonical_sha256":"badfa032d5c2f485c4aed7ba95996732c1e06897379e29d7d901ec9440c353e8","source":{"kind":"arxiv","id":"2607.14846","version":1},"attestation_state":"computed","paper":{"title":"RW-Voice-EQ Bench: A Real World Benchmark for Evaluating Voice AI Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SD","authors_text":"Alice Baird, David Ayllon, Franc Camps-Febrer, Georg Streich, Hoon Shin, Jakub Piotr C{\\l}apa, Jeffrey Brooks, Jens Madsen, Olya Ossipova, Panagiotis Tzirakis, Rashish Tandon, Sharath Rao, Theo Lebryk, Tigran Soghbatyan","submitted_at":"2026-07-16T11:15:16Z","abstract_excerpt":"Current voice AI benchmarks typically evaluate isolated capabilities such as speech intelligibility, word error rate, or text-based dialogue quality, but they rarely test whether systems harness the acoustic information that distinguishes spoken language from its textual representation. To this end, we introduce the Real World Voice EQ Bench, a multidimensional benchmark for evaluating voice AI across text-to-speech (TTS), speech-to-speech (STS), speech understanding (SU), and automatic speech recognition (ASR). Our evaluations indicate that performance is highly dimension-specific. For TTS, n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.14846","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2026-07-16T11:15:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1df6a3806bf7dfc920be46361b1dd5efe729f2b72a3c99a4cd170cc215ad09c3","abstract_canon_sha256":"8fbecab05ba4ca7804bbf457cb0e2ad7b810b8a1f2b731127bea6be041be32da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T01:21:34.223008Z","signature_b64":"587VL9At1hVe5HOJXYaiy+mq2E2E/CPX2rRS8rcqEputPYKcwEyJTxQ2eq+D9mV9FraUjG3qpEy7lcjsfd4yDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"badfa032d5c2f485c4aed7ba95996732c1e06897379e29d7d901ec9440c353e8","last_reissued_at":"2026-07-17T01:21:34.222143Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T01:21:34.222143Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RW-Voice-EQ Bench: A Real World Benchmark for Evaluating Voice AI Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SD","authors_text":"Alice Baird, David Ayllon, Franc Camps-Febrer, Georg Streich, Hoon Shin, Jakub Piotr C{\\l}apa, Jeffrey Brooks, Jens Madsen, Olya Ossipova, Panagiotis Tzirakis, Rashish Tandon, Sharath Rao, Theo Lebryk, Tigran Soghbatyan","submitted_at":"2026-07-16T11:15:16Z","abstract_excerpt":"Current voice AI benchmarks typically evaluate isolated capabilities such as speech intelligibility, word error rate, or text-based dialogue quality, but they rarely test whether systems harness the acoustic information that distinguishes spoken language from its textual representation. To this end, we introduce the Real World Voice EQ Bench, a multidimensional benchmark for evaluating voice AI across text-to-speech (TTS), speech-to-speech (STS), speech understanding (SU), and automatic speech recognition (ASR). Our evaluations indicate that performance is highly dimension-specific. For TTS, n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.14846","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.14846/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.14846","created_at":"2026-07-17T01:21:34.222580+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.14846v1","created_at":"2026-07-17T01:21:34.222580+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.14846","created_at":"2026-07-17T01:21:34.222580+00:00"},{"alias_kind":"pith_short_12","alias_value":"XLP2AMWVYL2I","created_at":"2026-07-17T01:21:34.222580+00:00"},{"alias_kind":"pith_short_16","alias_value":"XLP2AMWVYL2ILRFO","created_at":"2026-07-17T01:21:34.222580+00:00"},{"alias_kind":"pith_short_8","alias_value":"XLP2AMWV","created_at":"2026-07-17T01:21:34.222580+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL","json":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL.json","graph_json":"https://pith.science/api/pith-number/XLP2AMWVYL2ILRFO265JLGLHGL/graph.json","events_json":"https://pith.science/api/pith-number/XLP2AMWVYL2ILRFO265JLGLHGL/events.json","paper":"https://pith.science/paper/XLP2AMWV"},"agent_actions":{"view_html":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL","download_json":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL.json","view_paper":"https://pith.science/paper/XLP2AMWV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.14846&json=true","fetch_graph":"https://pith.science/api/pith-number/XLP2AMWVYL2ILRFO265JLGLHGL/graph.json","fetch_events":"https://pith.science/api/pith-number/XLP2AMWVYL2ILRFO265JLGLHGL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL/action/storage_attestation","attest_author":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL/action/author_attestation","sign_citation":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL/action/citation_signature","submit_replication":"https://pith.science/pith/XLP2AMWVYL2ILRFO265JLGLHGL/action/replication_record"}},"created_at":"2026-07-17T01:21:34.222580+00:00","updated_at":"2026-07-17T01:21:34.222580+00:00"}