{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QUDYJ5BCL52USGN4JIFVF2JMEP","short_pith_number":"pith:QUDYJ5BC","schema_version":"1.0","canonical_sha256":"850784f4225f754919bc4a0b52e92c23c0e745b1eaa72352a18b376460a33299","source":{"kind":"arxiv","id":"2306.14610","version":1},"attestation_state":"computed","paper":{"title":"SugarCrepe: Fixing Hackable Benchmarks for Vision-Language Compositionality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aniruddha Kembhavi, Cheng-Yu Hsieh, Jieyu Zhang, Ranjay Krishna, Zixian Ma","submitted_at":"2023-06-26T11:35:22Z","abstract_excerpt":"In the last year alone, a surge of new benchmarks to measure compositional understanding of vision-language models have permeated the machine learning ecosystem. Given an image, these benchmarks probe a model's ability to identify its associated caption amongst a set of compositional distractors. Surprisingly, we find significant biases in all these benchmarks rendering them hackable. This hackability is so dire that blind models with no access to the image outperform state-of-the-art vision-language models. To remedy this rampant vulnerability, we introduce SugarCrepe, a new benchmark for vis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.14610","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-26T11:35:22Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"aae4e2c6daa9d7284d8f2345e072abe4bde0a59ef6506fbfa08f7591c30181be","abstract_canon_sha256":"07826b186eb3ae16d275c68caaff35bec43b45ae2eb661e99a6f82f4794eb13c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:24:38.537553Z","signature_b64":"5LokadvDLN6mH0KNPY6gDWB6jtz9nnedM8Euy9i6Q3if7fP1Pf4NowlQtRml3N/keHUwMIz/zMdLOxWnD/KYAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"850784f4225f754919bc4a0b52e92c23c0e745b1eaa72352a18b376460a33299","last_reissued_at":"2026-07-05T06:24:38.537087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:24:38.537087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SugarCrepe: Fixing Hackable Benchmarks for Vision-Language Compositionality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aniruddha Kembhavi, Cheng-Yu Hsieh, Jieyu Zhang, Ranjay Krishna, Zixian Ma","submitted_at":"2023-06-26T11:35:22Z","abstract_excerpt":"In the last year alone, a surge of new benchmarks to measure compositional understanding of vision-language models have permeated the machine learning ecosystem. Given an image, these benchmarks probe a model's ability to identify its associated caption amongst a set of compositional distractors. Surprisingly, we find significant biases in all these benchmarks rendering them hackable. This hackability is so dire that blind models with no access to the image outperform state-of-the-art vision-language models. To remedy this rampant vulnerability, we introduce SugarCrepe, a new benchmark for vis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.14610","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.14610/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.14610","created_at":"2026-07-05T06:24:38.537149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.14610v1","created_at":"2026-07-05T06:24:38.537149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.14610","created_at":"2026-07-05T06:24:38.537149+00:00"},{"alias_kind":"pith_short_12","alias_value":"QUDYJ5BCL52U","created_at":"2026-07-05T06:24:38.537149+00:00"},{"alias_kind":"pith_short_16","alias_value":"QUDYJ5BCL52USGN4","created_at":"2026-07-05T06:24:38.537149+00:00"},{"alias_kind":"pith_short_8","alias_value":"QUDYJ5BC","created_at":"2026-07-05T06:24:38.537149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP","json":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP.json","graph_json":"https://pith.science/api/pith-number/QUDYJ5BCL52USGN4JIFVF2JMEP/graph.json","events_json":"https://pith.science/api/pith-number/QUDYJ5BCL52USGN4JIFVF2JMEP/events.json","paper":"https://pith.science/paper/QUDYJ5BC"},"agent_actions":{"view_html":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP","download_json":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP.json","view_paper":"https://pith.science/paper/QUDYJ5BC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.14610&json=true","fetch_graph":"https://pith.science/api/pith-number/QUDYJ5BCL52USGN4JIFVF2JMEP/graph.json","fetch_events":"https://pith.science/api/pith-number/QUDYJ5BCL52USGN4JIFVF2JMEP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP/action/storage_attestation","attest_author":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP/action/author_attestation","sign_citation":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP/action/citation_signature","submit_replication":"https://pith.science/pith/QUDYJ5BCL52USGN4JIFVF2JMEP/action/replication_record"}},"created_at":"2026-07-05T06:24:38.537149+00:00","updated_at":"2026-07-05T06:24:38.537149+00:00"}