{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UIQVDCFHYTNR5IJKHAF7RRFUVK","short_pith_number":"pith:UIQVDCFH","schema_version":"1.0","canonical_sha256":"a2215188a7c4db1ea12a380bf8c4b4aa81b741710dfbfdb9d9e622c76ce39c27","source":{"kind":"arxiv","id":"2406.08164","version":3},"attestation_state":"computed","paper":{"title":"ConMe: Rethinking Evaluation of Compositional Reasoning for Modern VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Assaf Arbelle, Aude Oliva, Chuang Gan, Hilde Kuehne, Irene Huang, Jacob A. Hansen, Leonid Karlinsky, M. Jehanzeb Mirza, Roei Herzig, Rogerio Feris, Sivan Doveh, Trevor Darrell, Victor Ion Butoi, Wei Lin","submitted_at":"2024-06-12T12:54:27Z","abstract_excerpt":"Compositional Reasoning (CR) entails grasping the significance of attributes, relations, and word order. Recent Vision-Language Models (VLMs), comprising a visual encoder and a Large Language Model (LLM) decoder, have demonstrated remarkable proficiency in such reasoning tasks. This prompts a crucial question: have VLMs effectively tackled the CR challenge? We conjecture that existing CR benchmarks may not adequately push the boundaries of modern VLMs due to the reliance on an LLM-only negative text generation pipeline. Consequently, the negatives produced either appear as outliers from the na"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08164","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-12T12:54:27Z","cross_cats_sorted":[],"title_canon_sha256":"70d850ce6a60884e06e2382cf3a8a90b5d6cc81026868432311d7c49731480e1","abstract_canon_sha256":"d7d13df13342a44d3c8b31a380545bed973a88a07a7f6c68a44154b68470ceca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:36.325559Z","signature_b64":"gs7d/BIlEOsntqJJ0crm/m5O3ZhW2knwx/CViNxEL7z2IhrhIfePau7jydb2EIWET4fGXE8prLqQKCPkxInuCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2215188a7c4db1ea12a380bf8c4b4aa81b741710dfbfdb9d9e622c76ce39c27","last_reissued_at":"2026-07-05T09:34:36.325061Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:36.325061Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ConMe: Rethinking Evaluation of Compositional Reasoning for Modern VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Assaf Arbelle, Aude Oliva, Chuang Gan, Hilde Kuehne, Irene Huang, Jacob A. Hansen, Leonid Karlinsky, M. Jehanzeb Mirza, Roei Herzig, Rogerio Feris, Sivan Doveh, Trevor Darrell, Victor Ion Butoi, Wei Lin","submitted_at":"2024-06-12T12:54:27Z","abstract_excerpt":"Compositional Reasoning (CR) entails grasping the significance of attributes, relations, and word order. Recent Vision-Language Models (VLMs), comprising a visual encoder and a Large Language Model (LLM) decoder, have demonstrated remarkable proficiency in such reasoning tasks. This prompts a crucial question: have VLMs effectively tackled the CR challenge? We conjecture that existing CR benchmarks may not adequately push the boundaries of modern VLMs due to the reliance on an LLM-only negative text generation pipeline. Consequently, the negatives produced either appear as outliers from the na"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08164","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08164/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08164","created_at":"2026-07-05T09:34:36.325122+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08164v3","created_at":"2026-07-05T09:34:36.325122+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08164","created_at":"2026-07-05T09:34:36.325122+00:00"},{"alias_kind":"pith_short_12","alias_value":"UIQVDCFHYTNR","created_at":"2026-07-05T09:34:36.325122+00:00"},{"alias_kind":"pith_short_16","alias_value":"UIQVDCFHYTNR5IJK","created_at":"2026-07-05T09:34:36.325122+00:00"},{"alias_kind":"pith_short_8","alias_value":"UIQVDCFH","created_at":"2026-07-05T09:34:36.325122+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04733","citing_title":"Discovering Failure Modes in Vision-Language Models using RL","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK","json":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK.json","graph_json":"https://pith.science/api/pith-number/UIQVDCFHYTNR5IJKHAF7RRFUVK/graph.json","events_json":"https://pith.science/api/pith-number/UIQVDCFHYTNR5IJKHAF7RRFUVK/events.json","paper":"https://pith.science/paper/UIQVDCFH"},"agent_actions":{"view_html":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK","download_json":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK.json","view_paper":"https://pith.science/paper/UIQVDCFH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08164&json=true","fetch_graph":"https://pith.science/api/pith-number/UIQVDCFHYTNR5IJKHAF7RRFUVK/graph.json","fetch_events":"https://pith.science/api/pith-number/UIQVDCFHYTNR5IJKHAF7RRFUVK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK/action/storage_attestation","attest_author":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK/action/author_attestation","sign_citation":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK/action/citation_signature","submit_replication":"https://pith.science/pith/UIQVDCFHYTNR5IJKHAF7RRFUVK/action/replication_record"}},"created_at":"2026-07-05T09:34:36.325122+00:00","updated_at":"2026-07-05T09:34:36.325122+00:00"}