{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3KWC3FY7WDKBQAMG3D67CRA5VE","short_pith_number":"pith:3KWC3FY7","schema_version":"1.0","canonical_sha256":"daac2d971fb0d4180186d8fdf1441da900ad95b65dd6358b4e84071d5d350260","source":{"kind":"arxiv","id":"2507.04699","version":1},"attestation_state":"computed","paper":{"title":"A Visual Leap in CLIP Compositionality Reasoning through Generation of Counterfactual Sets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuanwei Huang, Hongyan Fei, Jiapei Zhang, Jie Zhou, Jinchao Zhang, Yeshuang Zhu, Ying Deng, Zexi Jia, Zhiqiang Yuan","submitted_at":"2025-07-07T06:47:10Z","abstract_excerpt":"Vision-language models (VLMs) often struggle with compositional reasoning due to insufficient high-quality image-text data. To tackle this challenge, we propose a novel block-based diffusion approach that automatically generates counterfactual datasets without manual annotation. Our method utilizes large language models to identify entities and their spatial relationships. It then independently generates image blocks as \"puzzle pieces\" coherently arranged according to specified compositional rules. This process creates diverse, high-fidelity counterfactual image-text pairs with precisely contr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.04699","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-07T06:47:10Z","cross_cats_sorted":[],"title_canon_sha256":"9c46fa634e0212b9e346681974f6c65a6a72a94a620e6303bfc3482e6c38584e","abstract_canon_sha256":"4bc7a3398d71b5ea048831f03526a4fd71cf298d275968052dd1179cd630b1f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:52.347895Z","signature_b64":"w5jY7VtA9eMkYtAYJbQ3w0581y9dj2FSuX/nAT2Bd5owvn3XSa1qNG+oGnSalqCn4jGvxL+zbmjOqsaG1xs1Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daac2d971fb0d4180186d8fdf1441da900ad95b65dd6358b4e84071d5d350260","last_reissued_at":"2026-07-05T11:32:52.347450Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:52.347450Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Visual Leap in CLIP Compositionality Reasoning through Generation of Counterfactual Sets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuanwei Huang, Hongyan Fei, Jiapei Zhang, Jie Zhou, Jinchao Zhang, Yeshuang Zhu, Ying Deng, Zexi Jia, Zhiqiang Yuan","submitted_at":"2025-07-07T06:47:10Z","abstract_excerpt":"Vision-language models (VLMs) often struggle with compositional reasoning due to insufficient high-quality image-text data. To tackle this challenge, we propose a novel block-based diffusion approach that automatically generates counterfactual datasets without manual annotation. Our method utilizes large language models to identify entities and their spatial relationships. It then independently generates image blocks as \"puzzle pieces\" coherently arranged according to specified compositional rules. This process creates diverse, high-fidelity counterfactual image-text pairs with precisely contr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.04699","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.04699/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.04699","created_at":"2026-07-05T11:32:52.347524+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.04699v1","created_at":"2026-07-05T11:32:52.347524+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.04699","created_at":"2026-07-05T11:32:52.347524+00:00"},{"alias_kind":"pith_short_12","alias_value":"3KWC3FY7WDKB","created_at":"2026-07-05T11:32:52.347524+00:00"},{"alias_kind":"pith_short_16","alias_value":"3KWC3FY7WDKBQAMG","created_at":"2026-07-05T11:32:52.347524+00:00"},{"alias_kind":"pith_short_8","alias_value":"3KWC3FY7","created_at":"2026-07-05T11:32:52.347524+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE","json":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE.json","graph_json":"https://pith.science/api/pith-number/3KWC3FY7WDKBQAMG3D67CRA5VE/graph.json","events_json":"https://pith.science/api/pith-number/3KWC3FY7WDKBQAMG3D67CRA5VE/events.json","paper":"https://pith.science/paper/3KWC3FY7"},"agent_actions":{"view_html":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE","download_json":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE.json","view_paper":"https://pith.science/paper/3KWC3FY7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.04699&json=true","fetch_graph":"https://pith.science/api/pith-number/3KWC3FY7WDKBQAMG3D67CRA5VE/graph.json","fetch_events":"https://pith.science/api/pith-number/3KWC3FY7WDKBQAMG3D67CRA5VE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE/action/storage_attestation","attest_author":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE/action/author_attestation","sign_citation":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE/action/citation_signature","submit_replication":"https://pith.science/pith/3KWC3FY7WDKBQAMG3D67CRA5VE/action/replication_record"}},"created_at":"2026-07-05T11:32:52.347524+00:00","updated_at":"2026-07-05T11:32:52.347524+00:00"}