{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:XMZ5LSEKHLSYP7NZSFU3CDKKEY","short_pith_number":"pith:XMZ5LSEK","schema_version":"1.0","canonical_sha256":"bb33d5c88a3ae587fdb99169b10d4a2627ad5ff25d5fce9cdc5f4a55e7ddc1cf","source":{"kind":"arxiv","id":"2608.09772","version":1},"attestation_state":"computed","paper":{"title":"PragMatch: Separating Pragmatic Incongruity from Cross-Modal Mismatch in Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"2), (2) Max Planck Institute for Informatics, Germany, Germany), Varsha Suresh (2) ((1) Saarland University, Vera Demberg (1, Zhanna Mukhametsharip (1)","submitted_at":"2026-08-10T16:00:04Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have demonstrated strong performance on multimodal benchmarks, yet it remains unclear whether they genuinely reason about relationships between images and text or rely on superficial correlations, known as shortcut learning. This question is particularly important for multimodal sarcasm detection, where successful prediction depends on recognizing pragmatic incongruity rather than treating sarcasm as simple image-text mismatch. We introduce PragMatch, a controlled benchmark of 3,000 image-text pairs derived from MMSD2.0, including original sarcastic example"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.09772","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-08-10T16:00:04Z","cross_cats_sorted":[],"title_canon_sha256":"fd96ce4cc767d49f46649681d48d0d303cf912243efabb750d572ef488d2bd33","abstract_canon_sha256":"69cb03fc44af6f7f38e3c7c3e5d5957726518aa3f96c9d88dd1974b2b4ea7542"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T02:24:59.445561Z","signature_b64":"rE5dg6Y7TrcU/PZxtuLqRvfFjXLXneq1kH6sxi3udbkEdcr0T9bpLzn4HwxWtU6DOvHw/PBHcpXYfy/6xmPXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb33d5c88a3ae587fdb99169b10d4a2627ad5ff25d5fce9cdc5f4a55e7ddc1cf","last_reissued_at":"2026-08-11T02:24:59.443734Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T02:24:59.443734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PragMatch: Separating Pragmatic Incongruity from Cross-Modal Mismatch in Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"2), (2) Max Planck Institute for Informatics, Germany, Germany), Varsha Suresh (2) ((1) Saarland University, Vera Demberg (1, Zhanna Mukhametsharip (1)","submitted_at":"2026-08-10T16:00:04Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have demonstrated strong performance on multimodal benchmarks, yet it remains unclear whether they genuinely reason about relationships between images and text or rely on superficial correlations, known as shortcut learning. This question is particularly important for multimodal sarcasm detection, where successful prediction depends on recognizing pragmatic incongruity rather than treating sarcasm as simple image-text mismatch. We introduce PragMatch, a controlled benchmark of 3,000 image-text pairs derived from MMSD2.0, including original sarcastic example"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.09772","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.09772/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.09772","created_at":"2026-08-11T02:24:59.444417+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.09772v1","created_at":"2026-08-11T02:24:59.444417+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.09772","created_at":"2026-08-11T02:24:59.444417+00:00"},{"alias_kind":"pith_short_12","alias_value":"XMZ5LSEKHLSY","created_at":"2026-08-11T02:24:59.444417+00:00"},{"alias_kind":"pith_short_16","alias_value":"XMZ5LSEKHLSYP7NZ","created_at":"2026-08-11T02:24:59.444417+00:00"},{"alias_kind":"pith_short_8","alias_value":"XMZ5LSEK","created_at":"2026-08-11T02:24:59.444417+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY","json":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY.json","graph_json":"https://pith.science/api/pith-number/XMZ5LSEKHLSYP7NZSFU3CDKKEY/graph.json","events_json":"https://pith.science/api/pith-number/XMZ5LSEKHLSYP7NZSFU3CDKKEY/events.json","paper":"https://pith.science/paper/XMZ5LSEK"},"agent_actions":{"view_html":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY","download_json":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY.json","view_paper":"https://pith.science/paper/XMZ5LSEK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.09772&json=true","fetch_graph":"https://pith.science/api/pith-number/XMZ5LSEKHLSYP7NZSFU3CDKKEY/graph.json","fetch_events":"https://pith.science/api/pith-number/XMZ5LSEKHLSYP7NZSFU3CDKKEY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY/action/storage_attestation","attest_author":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY/action/author_attestation","sign_citation":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY/action/citation_signature","submit_replication":"https://pith.science/pith/XMZ5LSEKHLSYP7NZSFU3CDKKEY/action/replication_record"}},"created_at":"2026-08-11T02:24:59.444417+00:00","updated_at":"2026-08-11T02:24:59.444417+00:00"}