{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F4ANDLOL4GC4O6GIBCEBS46FNR","short_pith_number":"pith:F4ANDLOL","schema_version":"1.0","canonical_sha256":"2f00d1adcbe185c778c808881973c56c650a4e9c1eb91494a8c4b0411715eafd","source":{"kind":"arxiv","id":"2406.07546","version":2},"attestation_state":"computed","paper":{"title":"Commonsense-T2I Challenge: Can Text-to-Image Generation Models Understand Commonsense?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dan Roth, Muyu He, William Yang Wang, Xingyu Fu, Yujie Lu","submitted_at":"2024-06-11T17:59:48Z","abstract_excerpt":"We present a novel task and benchmark for evaluating the ability of text-to-image(T2I) generation models to produce images that align with commonsense in real life, which we call Commonsense-T2I. Given two adversarial text prompts containing an identical set of action words with minor differences, such as \"a lightbulb without electricity\" v.s. \"a lightbulb with electricity\", we evaluate whether T2I models can conduct visual-commonsense reasoning, e.g. produce images that fit \"the lightbulb is unlit\" vs. \"the lightbulb is lit\" correspondingly. Commonsense-T2I presents an adversarial challenge, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07546","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-11T17:59:48Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"533aa44eea37ff36c0932a0aa841140879b6f6500c467a880de985c91de6551c","abstract_canon_sha256":"f138ac53e5e519bf66f691603c1ce9aa197c69395a803210b7f73d4612fcedc1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:52.166430Z","signature_b64":"i8nd3vc1626EiEnCM5J8u8211Zo08TW/cynBbLJ8QkolKHRF6dGwAwvZtOMqCVC6nXVZXrdJD2EUfgOvxJRBDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f00d1adcbe185c778c808881973c56c650a4e9c1eb91494a8c4b0411715eafd","last_reissued_at":"2026-07-05T08:54:52.165918Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:52.165918Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Commonsense-T2I Challenge: Can Text-to-Image Generation Models Understand Commonsense?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dan Roth, Muyu He, William Yang Wang, Xingyu Fu, Yujie Lu","submitted_at":"2024-06-11T17:59:48Z","abstract_excerpt":"We present a novel task and benchmark for evaluating the ability of text-to-image(T2I) generation models to produce images that align with commonsense in real life, which we call Commonsense-T2I. Given two adversarial text prompts containing an identical set of action words with minor differences, such as \"a lightbulb without electricity\" v.s. \"a lightbulb with electricity\", we evaluate whether T2I models can conduct visual-commonsense reasoning, e.g. produce images that fit \"the lightbulb is unlit\" vs. \"the lightbulb is lit\" correspondingly. Commonsense-T2I presents an adversarial challenge, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07546","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07546/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07546","created_at":"2026-07-05T08:54:52.165973+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07546v2","created_at":"2026-07-05T08:54:52.165973+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07546","created_at":"2026-07-05T08:54:52.165973+00:00"},{"alias_kind":"pith_short_12","alias_value":"F4ANDLOL4GC4","created_at":"2026-07-05T08:54:52.165973+00:00"},{"alias_kind":"pith_short_16","alias_value":"F4ANDLOL4GC4O6GI","created_at":"2026-07-05T08:54:52.165973+00:00"},{"alias_kind":"pith_short_8","alias_value":"F4ANDLOL","created_at":"2026-07-05T08:54:52.165973+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26738","citing_title":"Do Image Editing Models Understand Lighting?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30262","citing_title":"Intermediate Text Representation Guided Text-to-Image Generation for Enhancing One-and-Only Alignment","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04300","citing_title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05363","citing_title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07265","citing_title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR","json":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR.json","graph_json":"https://pith.science/api/pith-number/F4ANDLOL4GC4O6GIBCEBS46FNR/graph.json","events_json":"https://pith.science/api/pith-number/F4ANDLOL4GC4O6GIBCEBS46FNR/events.json","paper":"https://pith.science/paper/F4ANDLOL"},"agent_actions":{"view_html":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR","download_json":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR.json","view_paper":"https://pith.science/paper/F4ANDLOL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07546&json=true","fetch_graph":"https://pith.science/api/pith-number/F4ANDLOL4GC4O6GIBCEBS46FNR/graph.json","fetch_events":"https://pith.science/api/pith-number/F4ANDLOL4GC4O6GIBCEBS46FNR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR/action/storage_attestation","attest_author":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR/action/author_attestation","sign_citation":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR/action/citation_signature","submit_replication":"https://pith.science/pith/F4ANDLOL4GC4O6GIBCEBS46FNR/action/replication_record"}},"created_at":"2026-07-05T08:54:52.165973+00:00","updated_at":"2026-07-05T08:54:52.165973+00:00"}