{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RNVJ3G2QPS3LGR5P2OB5GRONS4","short_pith_number":"pith:RNVJ3G2Q","schema_version":"1.0","canonical_sha256":"8b6a9d9b507cb6b347afd383d345cd971dfc91c3fcf8e1d551660a38fcf20ce3","source":{"kind":"arxiv","id":"2508.03654","version":1},"attestation_state":"computed","paper":{"title":"Can Large Vision-Language Models Understand Multimodal Sarcasm?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Liqiang Jing, Xinyu Wang, Yue Zhang","submitted_at":"2025-08-05T17:05:11Z","abstract_excerpt":"Sarcasm is a complex linguistic phenomenon that involves a disparity between literal and intended meanings, making it challenging for sentiment analysis and other emotion-sensitive tasks. While traditional sarcasm detection methods primarily focus on text, recent approaches have incorporated multimodal information. However, the application of Large Visual Language Models (LVLMs) in Multimodal Sarcasm Analysis (MSA) remains underexplored. In this paper, we evaluate LVLMs in MSA tasks, specifically focusing on Multimodal Sarcasm Detection and Multimodal Sarcasm Explanation. Through comprehensive"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.03654","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-05T17:05:11Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"dae2024e0163fc17cdaef60b5021afe00c093e0a05b1fb27f54fb992dd33ef1a","abstract_canon_sha256":"4442a5be2a7b76dc881a98793710de7be1594af5ff9379ec0bb5b02587966e3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:00.663491Z","signature_b64":"+h0VArzuMAFbEVTipYXRWc5u/S9lMlz5KdrcHyfMkrOnteIZwHWCsGsd2lKeVFeEsv3kAeoc6o8MkoWso4aUCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b6a9d9b507cb6b347afd383d345cd971dfc91c3fcf8e1d551660a38fcf20ce3","last_reissued_at":"2026-07-05T11:49:00.663000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:00.663000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Vision-Language Models Understand Multimodal Sarcasm?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Liqiang Jing, Xinyu Wang, Yue Zhang","submitted_at":"2025-08-05T17:05:11Z","abstract_excerpt":"Sarcasm is a complex linguistic phenomenon that involves a disparity between literal and intended meanings, making it challenging for sentiment analysis and other emotion-sensitive tasks. While traditional sarcasm detection methods primarily focus on text, recent approaches have incorporated multimodal information. However, the application of Large Visual Language Models (LVLMs) in Multimodal Sarcasm Analysis (MSA) remains underexplored. In this paper, we evaluate LVLMs in MSA tasks, specifically focusing on Multimodal Sarcasm Detection and Multimodal Sarcasm Explanation. Through comprehensive"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.03654","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.03654/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.03654","created_at":"2026-07-05T11:49:00.663062+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.03654v1","created_at":"2026-07-05T11:49:00.663062+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.03654","created_at":"2026-07-05T11:49:00.663062+00:00"},{"alias_kind":"pith_short_12","alias_value":"RNVJ3G2QPS3L","created_at":"2026-07-05T11:49:00.663062+00:00"},{"alias_kind":"pith_short_16","alias_value":"RNVJ3G2QPS3LGR5P","created_at":"2026-07-05T11:49:00.663062+00:00"},{"alias_kind":"pith_short_8","alias_value":"RNVJ3G2Q","created_at":"2026-07-05T11:49:00.663062+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.08879","citing_title":"GRASP: Grounded CoT Reasoning with Dual-Stage Optimization for Multimodal Sarcasm Target Identification","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4","json":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4.json","graph_json":"https://pith.science/api/pith-number/RNVJ3G2QPS3LGR5P2OB5GRONS4/graph.json","events_json":"https://pith.science/api/pith-number/RNVJ3G2QPS3LGR5P2OB5GRONS4/events.json","paper":"https://pith.science/paper/RNVJ3G2Q"},"agent_actions":{"view_html":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4","download_json":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4.json","view_paper":"https://pith.science/paper/RNVJ3G2Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.03654&json=true","fetch_graph":"https://pith.science/api/pith-number/RNVJ3G2QPS3LGR5P2OB5GRONS4/graph.json","fetch_events":"https://pith.science/api/pith-number/RNVJ3G2QPS3LGR5P2OB5GRONS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4/action/storage_attestation","attest_author":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4/action/author_attestation","sign_citation":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4/action/citation_signature","submit_replication":"https://pith.science/pith/RNVJ3G2QPS3LGR5P2OB5GRONS4/action/replication_record"}},"created_at":"2026-07-05T11:49:00.663062+00:00","updated_at":"2026-07-05T11:49:00.663062+00:00"}