{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:63ZW2OLCJC2CM5VZHRVP55GDZB","short_pith_number":"pith:63ZW2OLC","schema_version":"1.0","canonical_sha256":"f6f36d396248b42676b93c6afef4c3c864cbc5757dfa70f411420fd8813a2b9b","source":{"kind":"arxiv","id":"2507.19040","version":1},"attestation_state":"computed","paper":{"title":"FD-Bench: A Full-Duplex Benchmarking Pipeline Designed for Full Duplex Spoken Dialogue Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Bin Ma, Chongjia Ni, Dianwen Ng, Eng Siong Chng, Yi-Wen Chao, Yizhou Peng, Yukun Ma","submitted_at":"2025-07-25T07:51:22Z","abstract_excerpt":"Full-duplex spoken dialogue systems (FDSDS) enable more natural human-machine interactions by allowing real-time user interruptions and backchanneling, compared to traditional SDS that rely on turn-taking. However, existing benchmarks lack metrics for FD scenes, e.g., evaluating model performance during user interruptions. In this paper, we present a comprehensive FD benchmarking pipeline utilizing LLMs, TTS, and ASR to address this gap. It assesses FDSDS's ability to handle user interruptions, manage delays, and maintain robustness in challenging scenarios with diverse novel metrics. We appli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.19040","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-07-25T07:51:22Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"46287364d003a7d272fc0e51c9467cd3592842e048dd5782aa1a45e3779cc2a0","abstract_canon_sha256":"d563d4cc9387542e6059614a2005c58bf8efce9381c7d105a0556fb6239eefa9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:18.565036Z","signature_b64":"4dkh+pZ4iirlx/JiQ6QzBZyuK/88Wc9aQ7rZqC5YP1tsjlKm7sR8hHjcPhDiyKAvFfR9XyKAy6zBLENnIykqAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6f36d396248b42676b93c6afef4c3c864cbc5757dfa70f411420fd8813a2b9b","last_reissued_at":"2026-07-05T11:43:18.564441Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:18.564441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FD-Bench: A Full-Duplex Benchmarking Pipeline Designed for Full Duplex Spoken Dialogue Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Bin Ma, Chongjia Ni, Dianwen Ng, Eng Siong Chng, Yi-Wen Chao, Yizhou Peng, Yukun Ma","submitted_at":"2025-07-25T07:51:22Z","abstract_excerpt":"Full-duplex spoken dialogue systems (FDSDS) enable more natural human-machine interactions by allowing real-time user interruptions and backchanneling, compared to traditional SDS that rely on turn-taking. However, existing benchmarks lack metrics for FD scenes, e.g., evaluating model performance during user interruptions. In this paper, we present a comprehensive FD benchmarking pipeline utilizing LLMs, TTS, and ASR to address this gap. It assesses FDSDS's ability to handle user interruptions, manage delays, and maintain robustness in challenging scenarios with diverse novel metrics. We appli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.19040","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.19040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.19040","created_at":"2026-07-05T11:43:18.564511+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.19040v1","created_at":"2026-07-05T11:43:18.564511+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.19040","created_at":"2026-07-05T11:43:18.564511+00:00"},{"alias_kind":"pith_short_12","alias_value":"63ZW2OLCJC2C","created_at":"2026-07-05T11:43:18.564511+00:00"},{"alias_kind":"pith_short_16","alias_value":"63ZW2OLCJC2CM5VZ","created_at":"2026-07-05T11:43:18.564511+00:00"},{"alias_kind":"pith_short_8","alias_value":"63ZW2OLC","created_at":"2026-07-05T11:43:18.564511+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01804","citing_title":"SpeechEditBench: A Bilingual Multi-Attribute Benchmark for Instruction-Guided Speech Editing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30256","citing_title":"VideoFDB: Evaluating Full-Duplex Vision-Speech Capabilities in Conversational Agents","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20356","citing_title":"Synchronization and Turn-Taking in Full-Duplex Speech Dialogue Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01897","citing_title":"FastTurn: Unifying Acoustic and Streaming Semantic Cues for Low-Latency and Robust Turn Detection","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21406","citing_title":"Full-Duplex Interaction in Spoken Dialogue Systems: A Comprehensive Study from the ICASSP 2026 HumDial Challenge","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB","json":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB.json","graph_json":"https://pith.science/api/pith-number/63ZW2OLCJC2CM5VZHRVP55GDZB/graph.json","events_json":"https://pith.science/api/pith-number/63ZW2OLCJC2CM5VZHRVP55GDZB/events.json","paper":"https://pith.science/paper/63ZW2OLC"},"agent_actions":{"view_html":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB","download_json":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB.json","view_paper":"https://pith.science/paper/63ZW2OLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.19040&json=true","fetch_graph":"https://pith.science/api/pith-number/63ZW2OLCJC2CM5VZHRVP55GDZB/graph.json","fetch_events":"https://pith.science/api/pith-number/63ZW2OLCJC2CM5VZHRVP55GDZB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB/action/storage_attestation","attest_author":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB/action/author_attestation","sign_citation":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB/action/citation_signature","submit_replication":"https://pith.science/pith/63ZW2OLCJC2CM5VZHRVP55GDZB/action/replication_record"}},"created_at":"2026-07-05T11:43:18.564511+00:00","updated_at":"2026-07-05T11:43:18.564511+00:00"}