{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:FRAJGL7CDR4IZQSWF6CE365TIK","short_pith_number":"pith:FRAJGL7C","schema_version":"1.0","canonical_sha256":"2c40932fe21c788cc2562f844dfbb3429b68a874a5104c06babadf345a692891","source":{"kind":"arxiv","id":"2608.01119","version":1},"attestation_state":"computed","paper":{"title":"JoyAI-Talker: Full-Duplex Speech Interactive Large Model Built for Empathetic Voice Agents","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SD","authors_text":"Boya Dong, Chao Xue, Fan Yu, Hao Li, Hongfei Xue, Jiaqi Wang, Ji Miao, Jingdong Li, Jinming Chen, Lin Zhu, Ming Ke, Nan Duan, Ning Liu, Qi Wang, Tianyi Zhang, Wei Deng, Weisheng Han, Wenchao Wang, Xiangyu Liang, Yafeng Chen, Yankun Huang, Yinhao Bai, Yuan Liu, Yuan Zhang, Yu Gu, Yuqi Zhang, Yuxuan Wang, Zhangyu Xiao, Zhenfang Wang","submitted_at":"2026-08-02T09:27:51Z","abstract_excerpt":"We present JoyAI-Talker, a full-duplex speech dialogue system that delivers robust foundation model capabilities while empowering empathetic interaction and voice agent intelligence. JoyAI-Talker adopts a modular Thinker-Talker architecture and further implements a unified speech-text joint training pipeline to mitigate the common \"cognitive degradation\" bottleneck, thereby largely preserving the model's core textual reasoning, STEM, and logical capabilities while extending them to speech-based interaction. For expressive speech synthesis, the Talker module employs a text-controllable generati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.01119","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2026-08-02T09:27:51Z","cross_cats_sorted":[],"title_canon_sha256":"2d603a8ad4dae2e744eccce9c57d7978638209b836d9cc3bcc43e23c5e03e87b","abstract_canon_sha256":"6565399b118c5879f44549910d82e54d492f040907275b6287880f59778d96f2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T01:56:53.102981Z","signature_b64":"uKOYp2nhCG9uAyB2K49IQD24gtu7/RtfNfXd83ZkyLB3yuymkKt2ExLAF//ylr8vpuMCCqDl8kryI42gqvJlAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c40932fe21c788cc2562f844dfbb3429b68a874a5104c06babadf345a692891","last_reissued_at":"2026-08-04T01:56:53.101372Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T01:56:53.101372Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JoyAI-Talker: Full-Duplex Speech Interactive Large Model Built for Empathetic Voice Agents","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SD","authors_text":"Boya Dong, Chao Xue, Fan Yu, Hao Li, Hongfei Xue, Jiaqi Wang, Ji Miao, Jingdong Li, Jinming Chen, Lin Zhu, Ming Ke, Nan Duan, Ning Liu, Qi Wang, Tianyi Zhang, Wei Deng, Weisheng Han, Wenchao Wang, Xiangyu Liang, Yafeng Chen, Yankun Huang, Yinhao Bai, Yuan Liu, Yuan Zhang, Yu Gu, Yuqi Zhang, Yuxuan Wang, Zhangyu Xiao, Zhenfang Wang","submitted_at":"2026-08-02T09:27:51Z","abstract_excerpt":"We present JoyAI-Talker, a full-duplex speech dialogue system that delivers robust foundation model capabilities while empowering empathetic interaction and voice agent intelligence. JoyAI-Talker adopts a modular Thinker-Talker architecture and further implements a unified speech-text joint training pipeline to mitigate the common \"cognitive degradation\" bottleneck, thereby largely preserving the model's core textual reasoning, STEM, and logical capabilities while extending them to speech-based interaction. For expressive speech synthesis, the Talker module employs a text-controllable generati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.01119","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.01119/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.01119","created_at":"2026-08-04T01:56:53.102739+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.01119v1","created_at":"2026-08-04T01:56:53.102739+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.01119","created_at":"2026-08-04T01:56:53.102739+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRAJGL7CDR4I","created_at":"2026-08-04T01:56:53.102739+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRAJGL7CDR4IZQSW","created_at":"2026-08-04T01:56:53.102739+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRAJGL7C","created_at":"2026-08-04T01:56:53.102739+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK","json":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK.json","graph_json":"https://pith.science/api/pith-number/FRAJGL7CDR4IZQSWF6CE365TIK/graph.json","events_json":"https://pith.science/api/pith-number/FRAJGL7CDR4IZQSWF6CE365TIK/events.json","paper":"https://pith.science/paper/FRAJGL7C"},"agent_actions":{"view_html":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK","download_json":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK.json","view_paper":"https://pith.science/paper/FRAJGL7C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.01119&json=true","fetch_graph":"https://pith.science/api/pith-number/FRAJGL7CDR4IZQSWF6CE365TIK/graph.json","fetch_events":"https://pith.science/api/pith-number/FRAJGL7CDR4IZQSWF6CE365TIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK/action/storage_attestation","attest_author":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK/action/author_attestation","sign_citation":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK/action/citation_signature","submit_replication":"https://pith.science/pith/FRAJGL7CDR4IZQSWF6CE365TIK/action/replication_record"}},"created_at":"2026-08-04T01:56:53.102739+00:00","updated_at":"2026-08-04T01:56:53.102739+00:00"}