{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KKEHSO7CE5TH4QIJAZRFTOEVZ6","short_pith_number":"pith:KKEHSO7C","schema_version":"1.0","canonical_sha256":"5288793be227667e4109066259b895cf8c1591cee63df8ac42922b206fc90af5","source":{"kind":"arxiv","id":"2405.19487","version":2},"attestation_state":"computed","paper":{"title":"A Full-duplex Speech Dialogue Scheme Based On Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Peng Wang, Sijie Yan, Songshuo Lu, Wei Xia, Yaohua Tang, Yuanjun Xiong","submitted_at":"2024-05-29T20:05:46Z","abstract_excerpt":"We present a generative dialogue system capable of operating in a full-duplex manner, allowing for seamless interaction. It is based on a large language model (LLM) carefully aligned to be aware of a perception module, a motor function module, and the concept of a simple finite state machine (called neural FSM) with two states. The perception and motor function modules operate in tandem, allowing the system to speak and listen to the user simultaneously. The LLM generates textual tokens for inquiry responses and makes autonomous decisions to start responding to, wait for, or interrupt the user"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19487","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-29T20:05:46Z","cross_cats_sorted":[],"title_canon_sha256":"377ccb6affbf8643a457ed350a5e39bd076772f5fba9e9972b76cc89e8600f50","abstract_canon_sha256":"dccc543e4e0dc94666ceef34c81ef3873c6dc7ad970d5083d1ffcd0dbd49f224"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:54.702329Z","signature_b64":"386N1YuFmqtzffJeEdY8P5nTWMqMGQideVRv8SsRVdXbUlxsvkcyEMwbdoN4eYQ7vWvrmucIWE1daSxveWTIBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5288793be227667e4109066259b895cf8c1591cee63df8ac42922b206fc90af5","last_reissued_at":"2026-07-05T09:27:54.701752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:54.701752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Full-duplex Speech Dialogue Scheme Based On Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Peng Wang, Sijie Yan, Songshuo Lu, Wei Xia, Yaohua Tang, Yuanjun Xiong","submitted_at":"2024-05-29T20:05:46Z","abstract_excerpt":"We present a generative dialogue system capable of operating in a full-duplex manner, allowing for seamless interaction. It is based on a large language model (LLM) carefully aligned to be aware of a perception module, a motor function module, and the concept of a simple finite state machine (called neural FSM) with two states. The perception and motor function modules operate in tandem, allowing the system to speak and listen to the user simultaneously. The LLM generates textual tokens for inquiry responses and makes autonomous decisions to start responding to, wait for, or interrupt the user"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19487","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19487/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19487","created_at":"2026-07-05T09:27:54.701816+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19487v2","created_at":"2026-07-05T09:27:54.701816+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19487","created_at":"2026-07-05T09:27:54.701816+00:00"},{"alias_kind":"pith_short_12","alias_value":"KKEHSO7CE5TH","created_at":"2026-07-05T09:27:54.701816+00:00"},{"alias_kind":"pith_short_16","alias_value":"KKEHSO7CE5TH4QIJ","created_at":"2026-07-05T09:27:54.701816+00:00"},{"alias_kind":"pith_short_8","alias_value":"KKEHSO7C","created_at":"2026-07-05T09:27:54.701816+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02062","citing_title":"LMPAN: A Lightweight Multi-Path Alignment Network for Joint Full-Duplex Acoustic Echo Cancellation and Noise Suppression","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09186","citing_title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20755","citing_title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20755","citing_title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2410.00037","citing_title":"Moshi: a speech-text foundation model for real-time dialogue","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06765","citing_title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6","json":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6.json","graph_json":"https://pith.science/api/pith-number/KKEHSO7CE5TH4QIJAZRFTOEVZ6/graph.json","events_json":"https://pith.science/api/pith-number/KKEHSO7CE5TH4QIJAZRFTOEVZ6/events.json","paper":"https://pith.science/paper/KKEHSO7C"},"agent_actions":{"view_html":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6","download_json":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6.json","view_paper":"https://pith.science/paper/KKEHSO7C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19487&json=true","fetch_graph":"https://pith.science/api/pith-number/KKEHSO7CE5TH4QIJAZRFTOEVZ6/graph.json","fetch_events":"https://pith.science/api/pith-number/KKEHSO7CE5TH4QIJAZRFTOEVZ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6/action/storage_attestation","attest_author":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6/action/author_attestation","sign_citation":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6/action/citation_signature","submit_replication":"https://pith.science/pith/KKEHSO7CE5TH4QIJAZRFTOEVZ6/action/replication_record"}},"created_at":"2026-07-05T09:27:54.701816+00:00","updated_at":"2026-07-05T09:27:54.701816+00:00"}