{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:N5NXAVR4OWEDOROH6COBBM2YRG","short_pith_number":"pith:N5NXAVR4","schema_version":"1.0","canonical_sha256":"6f5b70563c75883745c7f09c10b35889981d341b874befb4253b209317e3bbfe","source":{"kind":"arxiv","id":"2408.16725","version":3},"attestation_state":"computed","paper":{"title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.AI","authors_text":"Changqiao Wu, Zhifei Xie","submitted_at":"2024-08-29T17:18:53Z","abstract_excerpt":"Recent advances in language models have achieved significant progress. GPT-4o, as a new milestone, has enabled real-time conversations with humans, demonstrating near-human natural fluency. Such human-computer interaction necessitates models with the capability to perform reasoning directly with the audio modality and generate output in streaming. However, this remains beyond the reach of current academic models, as they typically depend on extra TTS systems for speech synthesis, resulting in undesirable latency. This paper introduces the Mini-Omni, an audio-based end-to-end conversational mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.16725","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-08-29T17:18:53Z","cross_cats_sorted":["cs.CL","cs.HC","cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"24dffb17c87c05938e380dbff747987998b2a94d40b21809964635d755939569","abstract_canon_sha256":"2af68310aaba36bea4535adc8b1be03cd41de365c26ae3df6e0f0ef42b9dce18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:07.424241Z","signature_b64":"Pti9kVgrRVuIPE0WuJXDE+FTfAYu92/KpWVyrSViYLkoA++bhB7Iz2I11m899tWwhN9xDGv7J3yKHKmnZ6dDBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f5b70563c75883745c7f09c10b35889981d341b874befb4253b209317e3bbfe","last_reissued_at":"2026-07-05T09:31:07.423759Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:07.423759Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.AI","authors_text":"Changqiao Wu, Zhifei Xie","submitted_at":"2024-08-29T17:18:53Z","abstract_excerpt":"Recent advances in language models have achieved significant progress. GPT-4o, as a new milestone, has enabled real-time conversations with humans, demonstrating near-human natural fluency. Such human-computer interaction necessitates models with the capability to perform reasoning directly with the audio modality and generate output in streaming. However, this remains beyond the reach of current academic models, as they typically depend on extra TTS systems for speech synthesis, resulting in undesirable latency. This paper introduces the Mini-Omni, an audio-based end-to-end conversational mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.16725","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.16725/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.16725","created_at":"2026-07-05T09:31:07.423822+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.16725v3","created_at":"2026-07-05T09:31:07.423822+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.16725","created_at":"2026-07-05T09:31:07.423822+00:00"},{"alias_kind":"pith_short_12","alias_value":"N5NXAVR4OWED","created_at":"2026-07-05T09:31:07.423822+00:00"},{"alias_kind":"pith_short_16","alias_value":"N5NXAVR4OWEDOROH","created_at":"2026-07-05T09:31:07.423822+00:00"},{"alias_kind":"pith_short_8","alias_value":"N5NXAVR4","created_at":"2026-07-05T09:31:07.423822+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":39,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06540","citing_title":"Hierarchical Acoustic-Semantic Modeling: Modality Separation and Semantic Coherence for Full-Duplex SLMs","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05365","citing_title":"SPEARBench: A Benchmark for Naturalness Evaluation in Streaming Speech-to-Speech Language Models","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22473","citing_title":"Interleaved Speech Language Models Latently Work In Text","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21882","citing_title":"Streaming T5-based Text-to-Speech Synthesis with Limited Lookahead","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19453","citing_title":"A Survey of Full-Duplex Spoken Dialogue Systems: Architectural Hierarchy, Interaction Ontology, and Decision State Machine","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01345","citing_title":"TurnNat: Automatic Evaluation of Turn-Taking Naturalness in Dyadic Spoken Dialogue","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13544","citing_title":"Adaptive Turn-Taking for Real-time Multi-Party Voice Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02214","citing_title":"Unlocking Speech-Text Compositional Powers: Instruction-Following Speech Language Models without Instruction Tuning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12199","citing_title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11386","citing_title":"Overcoming State Inertia in Full-Duplex Spoken Language Models via Activation Steering","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08486","citing_title":"TRADE: Transducer-Augmented Decoder for Speech LLM","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00387","citing_title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01802","citing_title":"MOSS-Audio Technical Report","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30944","citing_title":"Preserving Speech-to-Text LLM Capabilities in Speech-to-Speech Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20755","citing_title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13544","citing_title":"Adaptive Turn-Taking for Real-time Multi-Party Voice Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27190","citing_title":"Learning When to Think While Listening in Large Audio-Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00851","citing_title":"Sympatheia: Emotionally Adaptive Voice Assistant with Continuous Affect Conditioning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01016","citing_title":"PolySpeech-100: A Large-Scale Benchmark for Speech Understanding Across 100+ Languages and Dialects","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20266","citing_title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20755","citing_title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21008","citing_title":"A Survey of Audio Reasoning in Multimodal Foundation Models","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2502.11946","citing_title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22220","citing_title":"StableToken: A Noise-Robust Semantic Speech Tokenizer for Resilient SpeechLLMs","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17196","citing_title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","ref_index":101,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG","json":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG.json","graph_json":"https://pith.science/api/pith-number/N5NXAVR4OWEDOROH6COBBM2YRG/graph.json","events_json":"https://pith.science/api/pith-number/N5NXAVR4OWEDOROH6COBBM2YRG/events.json","paper":"https://pith.science/paper/N5NXAVR4"},"agent_actions":{"view_html":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG","download_json":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG.json","view_paper":"https://pith.science/paper/N5NXAVR4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.16725&json=true","fetch_graph":"https://pith.science/api/pith-number/N5NXAVR4OWEDOROH6COBBM2YRG/graph.json","fetch_events":"https://pith.science/api/pith-number/N5NXAVR4OWEDOROH6COBBM2YRG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG/action/storage_attestation","attest_author":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG/action/author_attestation","sign_citation":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG/action/citation_signature","submit_replication":"https://pith.science/pith/N5NXAVR4OWEDOROH6COBBM2YRG/action/replication_record"}},"created_at":"2026-07-05T09:31:07.423822+00:00","updated_at":"2026-07-05T09:31:07.423822+00:00"}