{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4KAVDJNROWR7GXRDARH24BYRX2","short_pith_number":"pith:4KAVDJNR","schema_version":"1.0","canonical_sha256":"e28151a5b175a3f35e23044fae0711beb2af6794acd67363e2052f80e7ac261b","source":{"kind":"arxiv","id":"2506.13642","version":2},"attestation_state":"computed","paper":{"title":"Stream-Omni: Simultaneous Multimodal Interactions with Large Language-Vision-Speech Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.SD","eess.AS"],"primary_cat":"cs.AI","authors_text":"Qingkai Fang, Shaolei Zhang, Shoutao Guo, Yang Feng, Yan Zhou","submitted_at":"2025-06-16T16:06:45Z","abstract_excerpt":"The emergence of GPT-4o-like large multimodal models (LMMs) has raised the exploration of integrating text, vision, and speech modalities to support more flexible multimodal interaction. Existing LMMs typically concatenate representation of modalities along the sequence dimension and feed them into a large language model (LLM) backbone. While sequence-dimension concatenation is straightforward for modality integration, it often relies heavily on large-scale data to learn modality alignments. In this paper, we aim to model the relationships between modalities more purposefully, thereby achievin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13642","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-16T16:06:45Z","cross_cats_sorted":["cs.CL","cs.CV","cs.SD","eess.AS"],"title_canon_sha256":"bb8cb7c9666f27aa75702e1e0230c3f0e38d9081fb8e4fac92c943e707490235","abstract_canon_sha256":"c028a750390db9cf039bc87dbde5970c301929a56d5f3fcb002017c46f9fafa9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:12.814822Z","signature_b64":"2701nohOAS78nWiV4M1gYlVQ2YyIoWmJwgt3vk8eEyD6bwoBuCRMX/Gq91vP12sxsjq01xWaS+tpJb9hBfuVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e28151a5b175a3f35e23044fae0711beb2af6794acd67363e2052f80e7ac261b","last_reissued_at":"2026-07-05T11:25:12.814280Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:12.814280Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stream-Omni: Simultaneous Multimodal Interactions with Large Language-Vision-Speech Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.SD","eess.AS"],"primary_cat":"cs.AI","authors_text":"Qingkai Fang, Shaolei Zhang, Shoutao Guo, Yang Feng, Yan Zhou","submitted_at":"2025-06-16T16:06:45Z","abstract_excerpt":"The emergence of GPT-4o-like large multimodal models (LMMs) has raised the exploration of integrating text, vision, and speech modalities to support more flexible multimodal interaction. Existing LMMs typically concatenate representation of modalities along the sequence dimension and feed them into a large language model (LLM) backbone. While sequence-dimension concatenation is straightforward for modality integration, it often relies heavily on large-scale data to learn modality alignments. In this paper, we aim to model the relationships between modalities more purposefully, thereby achievin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13642","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13642/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13642","created_at":"2026-07-05T11:25:12.814347+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13642v2","created_at":"2026-07-05T11:25:12.814347+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13642","created_at":"2026-07-05T11:25:12.814347+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KAVDJNROWR7","created_at":"2026-07-05T11:25:12.814347+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KAVDJNROWR7GXRD","created_at":"2026-07-05T11:25:12.814347+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KAVDJNR","created_at":"2026-07-05T11:25:12.814347+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01667","citing_title":"Temporal and Cross-Modal Alignment for Enhanced Audiovisual Video Captioning","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02231","citing_title":"See, Hear, and Understand: Benchmarking Audiovisual Human Speech Understanding in Multimodal Large Language Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08000","citing_title":"PASK: Toward Intent-Aware Proactive Agents with Long-Term Memory","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2","json":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2.json","graph_json":"https://pith.science/api/pith-number/4KAVDJNROWR7GXRDARH24BYRX2/graph.json","events_json":"https://pith.science/api/pith-number/4KAVDJNROWR7GXRDARH24BYRX2/events.json","paper":"https://pith.science/paper/4KAVDJNR"},"agent_actions":{"view_html":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2","download_json":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2.json","view_paper":"https://pith.science/paper/4KAVDJNR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13642&json=true","fetch_graph":"https://pith.science/api/pith-number/4KAVDJNROWR7GXRDARH24BYRX2/graph.json","fetch_events":"https://pith.science/api/pith-number/4KAVDJNROWR7GXRDARH24BYRX2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2/action/storage_attestation","attest_author":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2/action/author_attestation","sign_citation":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2/action/citation_signature","submit_replication":"https://pith.science/pith/4KAVDJNROWR7GXRDARH24BYRX2/action/replication_record"}},"created_at":"2026-07-05T11:25:12.814347+00:00","updated_at":"2026-07-05T11:25:12.814347+00:00"}