{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V4VY77OHOZHSCND2DGLDPENMKH","short_pith_number":"pith:V4VY77OH","schema_version":"1.0","canonical_sha256":"af2b8ffdc7764f21347a19963791ac51d0fc91226cb90b41b3fe3513f30dc25e","source":{"kind":"arxiv","id":"2407.10488","version":1},"attestation_state":"computed","paper":{"title":"How and where does CLIP process negation?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Albert Gatt, Pablo Mosteiro, Vincent Quantmeyer","submitted_at":"2024-07-15T07:20:06Z","abstract_excerpt":"Various benchmarks have been proposed to test linguistic understanding in pre-trained vision \\& language (VL) models. Here we build on the existence task from the VALSE benchmark (Parcalabescu et al, 2022) which we use to test models' understanding of negation, a particularly interesting issue for multimodal models. However, while such VL benchmarks are useful for measuring model performance, they do not reveal anything about the internal processes through which these models arrive at their outputs in such visio-linguistic tasks. We take inspiration from the growing literature on model interpr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10488","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-15T07:20:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"55a557c655e95385e65ca67dd05ab13ecf0b0b391c625c34aa7fd8f09bbfc959","abstract_canon_sha256":"8d67221c583a0b94efbf2087ab2936af78208ffd6a1f9ad9018254a0cf508e13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:00.192517Z","signature_b64":"ynW0OrKfIGDGbSHfwjdc9xivd/hiOfWvWLj0BuwuL6Z2RR6WCuBrnh8qAfG8vHZnd20tSmmxtx3Qn1B3ATbKAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af2b8ffdc7764f21347a19963791ac51d0fc91226cb90b41b3fe3513f30dc25e","last_reissued_at":"2026-07-05T08:44:00.192175Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:00.192175Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How and where does CLIP process negation?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Albert Gatt, Pablo Mosteiro, Vincent Quantmeyer","submitted_at":"2024-07-15T07:20:06Z","abstract_excerpt":"Various benchmarks have been proposed to test linguistic understanding in pre-trained vision \\& language (VL) models. Here we build on the existence task from the VALSE benchmark (Parcalabescu et al, 2022) which we use to test models' understanding of negation, a particularly interesting issue for multimodal models. However, while such VL benchmarks are useful for measuring model performance, they do not reveal anything about the internal processes through which these models arrive at their outputs in such visio-linguistic tasks. We take inspiration from the growing literature on model interpr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10488","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10488/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10488","created_at":"2026-07-05T08:44:00.192230+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10488v1","created_at":"2026-07-05T08:44:00.192230+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10488","created_at":"2026-07-05T08:44:00.192230+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4VY77OHOZHS","created_at":"2026-07-05T08:44:00.192230+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4VY77OHOZHSCND2","created_at":"2026-07-05T08:44:00.192230+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4VY77OH","created_at":"2026-07-05T08:44:00.192230+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22946","citing_title":"NegVQA: Can Vision Language Models Understand Negation?","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH","json":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH.json","graph_json":"https://pith.science/api/pith-number/V4VY77OHOZHSCND2DGLDPENMKH/graph.json","events_json":"https://pith.science/api/pith-number/V4VY77OHOZHSCND2DGLDPENMKH/events.json","paper":"https://pith.science/paper/V4VY77OH"},"agent_actions":{"view_html":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH","download_json":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH.json","view_paper":"https://pith.science/paper/V4VY77OH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10488&json=true","fetch_graph":"https://pith.science/api/pith-number/V4VY77OHOZHSCND2DGLDPENMKH/graph.json","fetch_events":"https://pith.science/api/pith-number/V4VY77OHOZHSCND2DGLDPENMKH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH/action/storage_attestation","attest_author":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH/action/author_attestation","sign_citation":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH/action/citation_signature","submit_replication":"https://pith.science/pith/V4VY77OHOZHSCND2DGLDPENMKH/action/replication_record"}},"created_at":"2026-07-05T08:44:00.192230+00:00","updated_at":"2026-07-05T08:44:00.192230+00:00"}