{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ASD43CSNMS7OII2V2CYJ26UO44","short_pith_number":"pith:ASD43CSN","schema_version":"1.0","canonical_sha256":"0487cd8a4d64bee42355d0b09d7a8ee718f1e600f49c05f764d92645096e60a5","source":{"kind":"arxiv","id":"2502.04976","version":1},"attestation_state":"computed","paper":{"title":"Towards Multimodal Empathetic Response Generation: A Rich Text-Speech-Vision Avatar-based Benchmark","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.MM","authors_text":"Erik Cambria, Han Zhang, Hao Fei, Hong Han, Lizi Liao, Meng Luo, Zixiang Meng","submitted_at":"2025-02-07T14:50:10Z","abstract_excerpt":"Empathetic Response Generation (ERG) is one of the key tasks of the affective computing area, which aims to produce emotionally nuanced and compassionate responses to user's queries. However, existing ERG research is predominantly confined to the singleton text modality, limiting its effectiveness since human emotions are inherently conveyed through multiple modalities. To combat this, we introduce an avatar-based Multimodal ERG (MERG) task, entailing rich text, speech, and facial vision information. We first present a large-scale high-quality benchmark dataset, \\textbf{AvaMERG}, which extends"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04976","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.MM","submitted_at":"2025-02-07T14:50:10Z","cross_cats_sorted":[],"title_canon_sha256":"717f66d4fb8547c21cc268df228687afa50e87462d577ed9cbb284d06d441d03","abstract_canon_sha256":"dfc85307c64d0780aee13b380c12a136e94d735df3dc89a2af74b6c3bfcca234"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:00.731115Z","signature_b64":"EKXp7zuQ4gkv7hKDkob5PFgreBQZ0NsKVKisp4+XA9P68n9AF6MkHeOYUnLky07luXr8cTYwescc/GSCu/+KAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0487cd8a4d64bee42355d0b09d7a8ee718f1e600f49c05f764d92645096e60a5","last_reissued_at":"2026-07-05T10:11:00.730659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:00.730659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Multimodal Empathetic Response Generation: A Rich Text-Speech-Vision Avatar-based Benchmark","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.MM","authors_text":"Erik Cambria, Han Zhang, Hao Fei, Hong Han, Lizi Liao, Meng Luo, Zixiang Meng","submitted_at":"2025-02-07T14:50:10Z","abstract_excerpt":"Empathetic Response Generation (ERG) is one of the key tasks of the affective computing area, which aims to produce emotionally nuanced and compassionate responses to user's queries. However, existing ERG research is predominantly confined to the singleton text modality, limiting its effectiveness since human emotions are inherently conveyed through multiple modalities. To combat this, we introduce an avatar-based Multimodal ERG (MERG) task, entailing rich text, speech, and facial vision information. We first present a large-scale high-quality benchmark dataset, \\textbf{AvaMERG}, which extends"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04976","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04976/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04976","created_at":"2026-07-05T10:11:00.730719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04976v1","created_at":"2026-07-05T10:11:00.730719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04976","created_at":"2026-07-05T10:11:00.730719+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASD43CSNMS7O","created_at":"2026-07-05T10:11:00.730719+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASD43CSNMS7OII2V","created_at":"2026-07-05T10:11:00.730719+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASD43CSN","created_at":"2026-07-05T10:11:00.730719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":128,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18988","citing_title":"A Multi-Agent Framework with Structured Reasoning and Reflective Refinement for Multimodal Empathetic Response Generation","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44","json":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44.json","graph_json":"https://pith.science/api/pith-number/ASD43CSNMS7OII2V2CYJ26UO44/graph.json","events_json":"https://pith.science/api/pith-number/ASD43CSNMS7OII2V2CYJ26UO44/events.json","paper":"https://pith.science/paper/ASD43CSN"},"agent_actions":{"view_html":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44","download_json":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44.json","view_paper":"https://pith.science/paper/ASD43CSN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04976&json=true","fetch_graph":"https://pith.science/api/pith-number/ASD43CSNMS7OII2V2CYJ26UO44/graph.json","fetch_events":"https://pith.science/api/pith-number/ASD43CSNMS7OII2V2CYJ26UO44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44/action/storage_attestation","attest_author":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44/action/author_attestation","sign_citation":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44/action/citation_signature","submit_replication":"https://pith.science/pith/ASD43CSNMS7OII2V2CYJ26UO44/action/replication_record"}},"created_at":"2026-07-05T10:11:00.730719+00:00","updated_at":"2026-07-05T10:11:00.730719+00:00"}