{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RBLIOIEDSRXLXW4CZZP73GQ775","short_pith_number":"pith:RBLIOIED","schema_version":"1.0","canonical_sha256":"8856872083946ebbdb82ce5ffd9a1fff4203bba93978b7dc095a3ce4dc6a47ad","source":{"kind":"arxiv","id":"2406.03872","version":1},"attestation_state":"computed","paper":{"title":"BLSP-Emo: Towards Empathetic Large Speech-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chengqing Zong, Chen Wang, Jiajun Zhang, Junhong Wu, Minpeng Liao, Zhongqiang Huang","submitted_at":"2024-06-06T09:02:31Z","abstract_excerpt":"The recent release of GPT-4o showcased the potential of end-to-end multimodal models, not just in terms of low latency but also in their ability to understand and generate expressive speech with rich emotions. While the details are unknown to the open research community, it likely involves significant amounts of curated data and compute, neither of which is readily accessible. In this paper, we present BLSP-Emo (Bootstrapped Language-Speech Pretraining with Emotion support), a novel approach to developing an end-to-end speech-language model capable of understanding both semantics and emotions "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03872","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-06T09:02:31Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"adeeb450773e873cbb91d6fdc086d0fb38c07d5dcf752f717fc1dbcbbdaead96","abstract_canon_sha256":"4838fece1f825c73da8760da0cb07db168eede93c7e61b073822431df10248aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:17.204200Z","signature_b64":"4vOw8pkYZe3iwOIdOtVWP2CsiYETv6UnubrNCd0RNH+HjFaHVRFKwM2VIuxqA/jR5EGCGRk0sIFn2usEy4BFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8856872083946ebbdb82ce5ffd9a1fff4203bba93978b7dc095a3ce4dc6a47ad","last_reissued_at":"2026-07-05T08:28:17.203777Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:17.203777Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BLSP-Emo: Towards Empathetic Large Speech-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chengqing Zong, Chen Wang, Jiajun Zhang, Junhong Wu, Minpeng Liao, Zhongqiang Huang","submitted_at":"2024-06-06T09:02:31Z","abstract_excerpt":"The recent release of GPT-4o showcased the potential of end-to-end multimodal models, not just in terms of low latency but also in their ability to understand and generate expressive speech with rich emotions. While the details are unknown to the open research community, it likely involves significant amounts of curated data and compute, neither of which is readily accessible. In this paper, we present BLSP-Emo (Bootstrapped Language-Speech Pretraining with Emotion support), a novel approach to developing an end-to-end speech-language model capable of understanding both semantics and emotions "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03872","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03872","created_at":"2026-07-05T08:28:17.203833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03872v1","created_at":"2026-07-05T08:28:17.203833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03872","created_at":"2026-07-05T08:28:17.203833+00:00"},{"alias_kind":"pith_short_12","alias_value":"RBLIOIEDSRXL","created_at":"2026-07-05T08:28:17.203833+00:00"},{"alias_kind":"pith_short_16","alias_value":"RBLIOIEDSRXLXW4C","created_at":"2026-07-05T08:28:17.203833+00:00"},{"alias_kind":"pith_short_8","alias_value":"RBLIOIED","created_at":"2026-07-05T08:28:17.203833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11098","citing_title":"AffectCodec: Emotion-Preserving Neural Speech Codec for Expressive Speech Modeling","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775","json":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775.json","graph_json":"https://pith.science/api/pith-number/RBLIOIEDSRXLXW4CZZP73GQ775/graph.json","events_json":"https://pith.science/api/pith-number/RBLIOIEDSRXLXW4CZZP73GQ775/events.json","paper":"https://pith.science/paper/RBLIOIED"},"agent_actions":{"view_html":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775","download_json":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775.json","view_paper":"https://pith.science/paper/RBLIOIED","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03872&json=true","fetch_graph":"https://pith.science/api/pith-number/RBLIOIEDSRXLXW4CZZP73GQ775/graph.json","fetch_events":"https://pith.science/api/pith-number/RBLIOIEDSRXLXW4CZZP73GQ775/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775/action/storage_attestation","attest_author":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775/action/author_attestation","sign_citation":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775/action/citation_signature","submit_replication":"https://pith.science/pith/RBLIOIEDSRXLXW4CZZP73GQ775/action/replication_record"}},"created_at":"2026-07-05T08:28:17.203833+00:00","updated_at":"2026-07-05T08:28:17.203833+00:00"}