{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F54ELMDCE3J7P6IHAPM7H3F2MW","short_pith_number":"pith:F54ELMDC","schema_version":"1.0","canonical_sha256":"2f7845b06226d3f7f90703d9f3ecba65bf16613f5603157db6cad4e4bb10639f","source":{"kind":"arxiv","id":"2508.06564","version":2},"attestation_state":"computed","paper":{"title":"Grounding Emotion Recognition with Visual Prototypes: VEGA -- Revisiting CLIP in MERC","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dimitrios Kollias, Guanyu Hu, Xinyu Yang","submitted_at":"2025-08-06T19:43:58Z","abstract_excerpt":"Multimodal Emotion Recognition in Conversations remains a challenging task due to the complex interplay of textual, acoustic and visual signals. While recent models have improved performance via advanced fusion strategies, they often lack psychologically meaningful priors to guide multimodal alignment. In this paper, we revisit the use of CLIP and propose a novel Visual Emotion Guided Anchoring (VEGA) mechanism that introduces class-level visual semantics into the fusion and classification process. Distinct from prior work that primarily utilizes CLIP's textual encoder, our approach leverages "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.06564","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-06T19:43:58Z","cross_cats_sorted":[],"title_canon_sha256":"1eca760be197f2d0ab8f09ed8b7a32d74dbf0201d1bae4c72044fe3a662a0282","abstract_canon_sha256":"33a18e2ff0fe15e124a272756eb6cbfb47dff34b81449e494e27855746f076e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:07.225967Z","signature_b64":"nCMK4hTACrDCCW0kGAU3zGQWC0PKcwD9Piq/2TWXGTrMxRVcfDBla9/JMnvbGugzMRRu07KnnAQTu0jEJEU4DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f7845b06226d3f7f90703d9f3ecba65bf16613f5603157db6cad4e4bb10639f","last_reissued_at":"2026-07-05T11:53:07.225408Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:07.225408Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grounding Emotion Recognition with Visual Prototypes: VEGA -- Revisiting CLIP in MERC","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dimitrios Kollias, Guanyu Hu, Xinyu Yang","submitted_at":"2025-08-06T19:43:58Z","abstract_excerpt":"Multimodal Emotion Recognition in Conversations remains a challenging task due to the complex interplay of textual, acoustic and visual signals. While recent models have improved performance via advanced fusion strategies, they often lack psychologically meaningful priors to guide multimodal alignment. In this paper, we revisit the use of CLIP and propose a novel Visual Emotion Guided Anchoring (VEGA) mechanism that introduces class-level visual semantics into the fusion and classification process. Distinct from prior work that primarily utilizes CLIP's textual encoder, our approach leverages "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.06564","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.06564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.06564","created_at":"2026-07-05T11:53:07.225475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.06564v2","created_at":"2026-07-05T11:53:07.225475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.06564","created_at":"2026-07-05T11:53:07.225475+00:00"},{"alias_kind":"pith_short_12","alias_value":"F54ELMDCE3J7","created_at":"2026-07-05T11:53:07.225475+00:00"},{"alias_kind":"pith_short_16","alias_value":"F54ELMDCE3J7P6IH","created_at":"2026-07-05T11:53:07.225475+00:00"},{"alias_kind":"pith_short_8","alias_value":"F54ELMDC","created_at":"2026-07-05T11:53:07.225475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.12554","citing_title":"EmoVerse: A MLLMs-Driven Emotion Representation Dataset for Interpretable Visual Emotion Analysis","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW","json":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW.json","graph_json":"https://pith.science/api/pith-number/F54ELMDCE3J7P6IHAPM7H3F2MW/graph.json","events_json":"https://pith.science/api/pith-number/F54ELMDCE3J7P6IHAPM7H3F2MW/events.json","paper":"https://pith.science/paper/F54ELMDC"},"agent_actions":{"view_html":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW","download_json":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW.json","view_paper":"https://pith.science/paper/F54ELMDC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.06564&json=true","fetch_graph":"https://pith.science/api/pith-number/F54ELMDCE3J7P6IHAPM7H3F2MW/graph.json","fetch_events":"https://pith.science/api/pith-number/F54ELMDCE3J7P6IHAPM7H3F2MW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW/action/storage_attestation","attest_author":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW/action/author_attestation","sign_citation":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW/action/citation_signature","submit_replication":"https://pith.science/pith/F54ELMDCE3J7P6IHAPM7H3F2MW/action/replication_record"}},"created_at":"2026-07-05T11:53:07.225475+00:00","updated_at":"2026-07-05T11:53:07.225475+00:00"}