{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UKXVM6GZAX5CGTS2BABJTA4KTU","short_pith_number":"pith:UKXVM6GZ","schema_version":"1.0","canonical_sha256":"a2af5678d905fa234e5a080299838a9d212a4d569ce1a9b470dcb6e907ee3e72","source":{"kind":"arxiv","id":"2404.01291","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Text-to-Visual Generation with Image-to-Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Baiqi Li, Deepak Pathak, Deva Ramanan, Graham Neubig, Jiayao Li, Pengchuan Zhang, Xide Xia, Zhiqiu Lin","submitted_at":"2024-04-01T17:58:06Z","abstract_excerpt":"Despite significant progress in generative AI, comprehensive evaluation remains challenging because of the lack of effective metrics and standardized benchmarks. For instance, the widely-used CLIPScore measures the alignment between a (generated) image and text prompt, but it fails to produce reliable scores for complex prompts involving compositions of objects, attributes, and relations. One reason is that text encoders of CLIP can notoriously act as a \"bag of words\", conflating prompts such as \"the horse is eating the grass\" with \"the grass is eating the horse\". To address this, we introduce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.01291","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-01T17:58:06Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","cs.MM"],"title_canon_sha256":"4cd073d29a2790c1b795053c5ea9c08a7ff9f1f50f7bd23a82a73167a7c8af5b","abstract_canon_sha256":"7225e4e0bf457b5c9b8fbd5dee38e05774a32211de5f7d42ac2416b6103978ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:38.318642Z","signature_b64":"2jeRBZ46J6+X+kLwURVug+jTk6SEOSLiBYkA6FYBOwhbCaQlzbl6fINTgR42Rzo06/hxT+zmY/QotQFU9J4oDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2af5678d905fa234e5a080299838a9d212a4d569ce1a9b470dcb6e907ee3e72","last_reissued_at":"2026-07-05T08:33:38.318130Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:38.318130Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Text-to-Visual Generation with Image-to-Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Baiqi Li, Deepak Pathak, Deva Ramanan, Graham Neubig, Jiayao Li, Pengchuan Zhang, Xide Xia, Zhiqiu Lin","submitted_at":"2024-04-01T17:58:06Z","abstract_excerpt":"Despite significant progress in generative AI, comprehensive evaluation remains challenging because of the lack of effective metrics and standardized benchmarks. For instance, the widely-used CLIPScore measures the alignment between a (generated) image and text prompt, but it fails to produce reliable scores for complex prompts involving compositions of objects, attributes, and relations. One reason is that text encoders of CLIP can notoriously act as a \"bag of words\", conflating prompts such as \"the horse is eating the grass\" with \"the grass is eating the horse\". To address this, we introduce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01291","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.01291","created_at":"2026-07-05T08:33:38.318188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.01291v2","created_at":"2026-07-05T08:33:38.318188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01291","created_at":"2026-07-05T08:33:38.318188+00:00"},{"alias_kind":"pith_short_12","alias_value":"UKXVM6GZAX5C","created_at":"2026-07-05T08:33:38.318188+00:00"},{"alias_kind":"pith_short_16","alias_value":"UKXVM6GZAX5CGTS2","created_at":"2026-07-05T08:33:38.318188+00:00"},{"alias_kind":"pith_short_8","alias_value":"UKXVM6GZ","created_at":"2026-07-05T08:33:38.318188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06445","citing_title":"Analysis-by-Proxy: Localization Signals in VLMs Operating as Condition Encoders","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09076","citing_title":"Z-Reward: Beyond Scalar Rewards by Internalizing Reasoning into Score Distributions","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05268","citing_title":"Aggregating LLM-Based Weak Verifiers for Spatial Layout Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01803","citing_title":"OctoT2I: A Self-Evolving Agentic Text-to-Image Router","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02521","citing_title":"Drifting Preference Optimization for One-Step Generative Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17171","citing_title":"FireScope: Wildfire Risk Raster Prediction with a Chain-of-Thought Oracle","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22996","citing_title":"CoMoGen: COntrollable MOtion Dynamics and Interactions with Mask-Guided Video GENeration","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22469","citing_title":"MaSC: A Masked Similarity Metric for Evaluating Concept-Driven Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2406.03520","citing_title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05363","citing_title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2511.07756","citing_title":"Determinism of Randomness: Prompt-Residual Seed Shaping for Diffusion Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17171","citing_title":"FireScope: Wildfire Risk Raster Prediction with a Chain-of-Thought Oracle","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07703","citing_title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05472","citing_title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2602.03342","citing_title":"Tiled Prompts: Overcoming Prompt Misguidance in Image and Video Super-Resolution","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2409.04429","citing_title":"VILA-U: a Unified Foundation Model Integrating Visual Understanding and Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07265","citing_title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09860","citing_title":"RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00799","citing_title":"Multimodal Language Models Cannot Spot Spatial Inconsistencies","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":75,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU","json":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU.json","graph_json":"https://pith.science/api/pith-number/UKXVM6GZAX5CGTS2BABJTA4KTU/graph.json","events_json":"https://pith.science/api/pith-number/UKXVM6GZAX5CGTS2BABJTA4KTU/events.json","paper":"https://pith.science/paper/UKXVM6GZ"},"agent_actions":{"view_html":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU","download_json":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU.json","view_paper":"https://pith.science/paper/UKXVM6GZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.01291&json=true","fetch_graph":"https://pith.science/api/pith-number/UKXVM6GZAX5CGTS2BABJTA4KTU/graph.json","fetch_events":"https://pith.science/api/pith-number/UKXVM6GZAX5CGTS2BABJTA4KTU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU/action/storage_attestation","attest_author":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU/action/author_attestation","sign_citation":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU/action/citation_signature","submit_replication":"https://pith.science/pith/UKXVM6GZAX5CGTS2BABJTA4KTU/action/replication_record"}},"created_at":"2026-07-05T08:33:38.318188+00:00","updated_at":"2026-07-05T08:33:38.318188+00:00"}