{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:26CCG7ZKZW44TZPVCJ5VN5QPJD","short_pith_number":"pith:26CCG7ZK","schema_version":"1.0","canonical_sha256":"d784237f2acdb9c9e5f5127b56f60f48fbe97e1f34d9cf835c937a05a30e6267","source":{"kind":"arxiv","id":"2304.12995","version":1},"attestation_state":"computed","paper":{"title":"AudioGPT: Understanding and Generating Speech, Music, Sound, and Talking Head","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Dongchao Yang, Jiatong Shi, Jiawei Huang, Jinglin Liu, Mingze Li, Rongjie Huang, Shinji Watanabe, Xuankai Chang, Yi Ren, Yuning Wu, Zhenhui Ye, Zhiqing Hong, Zhou Zhao","submitted_at":"2023-04-25T17:05:38Z","abstract_excerpt":"Large language models (LLMs) have exhibited remarkable capabilities across a variety of domains and tasks, challenging our understanding of learning and cognition. Despite the recent success, current LLMs are not capable of processing complex audio information or conducting spoken conversations (like Siri or Alexa). In this work, we propose a multi-modal AI system named AudioGPT, which complements LLMs (i.e., ChatGPT) with 1) foundation models to process complex audio information and solve numerous understanding and generation tasks; and 2) the input/output interface (ASR, TTS) to support spok"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.12995","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-25T17:05:38Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"b4d0445ca42ae2b89619c773a0371124d3464f7392c760e0ac6a085fe76eba01","abstract_canon_sha256":"2288dcf42599ffeb7f9885df0b92a7bcaad80d0b89ae7c54233c299d2e7fb38e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:04:23.207979Z","signature_b64":"aB8EUNj84cIWj0vu/YNERf+uZsLw4VQtyBkRCF3fl/GNKTNb8NQNnczT7GHcQP8WpEghZrBbqpz/L9fVahYGCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d784237f2acdb9c9e5f5127b56f60f48fbe97e1f34d9cf835c937a05a30e6267","last_reissued_at":"2026-07-05T06:04:23.207573Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:04:23.207573Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AudioGPT: Understanding and Generating Speech, Music, Sound, and Talking Head","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Dongchao Yang, Jiatong Shi, Jiawei Huang, Jinglin Liu, Mingze Li, Rongjie Huang, Shinji Watanabe, Xuankai Chang, Yi Ren, Yuning Wu, Zhenhui Ye, Zhiqing Hong, Zhou Zhao","submitted_at":"2023-04-25T17:05:38Z","abstract_excerpt":"Large language models (LLMs) have exhibited remarkable capabilities across a variety of domains and tasks, challenging our understanding of learning and cognition. Despite the recent success, current LLMs are not capable of processing complex audio information or conducting spoken conversations (like Siri or Alexa). In this work, we propose a multi-modal AI system named AudioGPT, which complements LLMs (i.e., ChatGPT) with 1) foundation models to process complex audio information and solve numerous understanding and generation tasks; and 2) the input/output interface (ASR, TTS) to support spok"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.12995","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.12995/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.12995","created_at":"2026-07-05T06:04:23.207634+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.12995v1","created_at":"2026-07-05T06:04:23.207634+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.12995","created_at":"2026-07-05T06:04:23.207634+00:00"},{"alias_kind":"pith_short_12","alias_value":"26CCG7ZKZW44","created_at":"2026-07-05T06:04:23.207634+00:00"},{"alias_kind":"pith_short_16","alias_value":"26CCG7ZKZW44TZPV","created_at":"2026-07-05T06:04:23.207634+00:00"},{"alias_kind":"pith_short_8","alias_value":"26CCG7ZK","created_at":"2026-07-05T06:04:23.207634+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17152","citing_title":"Multilingual and Multimodal LLMs in the Wild: Building for Low-Resource Languages","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2310.13289","citing_title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02954","citing_title":"The World is Not Mono: Enabling Spatial Understanding in Large Audio-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2304.12244","citing_title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07919","citing_title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":294,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD","json":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD.json","graph_json":"https://pith.science/api/pith-number/26CCG7ZKZW44TZPVCJ5VN5QPJD/graph.json","events_json":"https://pith.science/api/pith-number/26CCG7ZKZW44TZPVCJ5VN5QPJD/events.json","paper":"https://pith.science/paper/26CCG7ZK"},"agent_actions":{"view_html":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD","download_json":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD.json","view_paper":"https://pith.science/paper/26CCG7ZK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.12995&json=true","fetch_graph":"https://pith.science/api/pith-number/26CCG7ZKZW44TZPVCJ5VN5QPJD/graph.json","fetch_events":"https://pith.science/api/pith-number/26CCG7ZKZW44TZPVCJ5VN5QPJD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD/action/storage_attestation","attest_author":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD/action/author_attestation","sign_citation":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD/action/citation_signature","submit_replication":"https://pith.science/pith/26CCG7ZKZW44TZPVCJ5VN5QPJD/action/replication_record"}},"created_at":"2026-07-05T06:04:23.207634+00:00","updated_at":"2026-07-05T06:04:23.207634+00:00"}