{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QCZDWQVCNNYS4V6QJCE3L7Q5UC","short_pith_number":"pith:QCZDWQVC","schema_version":"1.0","canonical_sha256":"80b23b42a26b712e57d04889b5fe1da08ac9ff00aef8ff179d435e145cda1613","source":{"kind":"arxiv","id":"2411.14062","version":2},"attestation_state":"computed","paper":{"title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hailang Huang, HuaQiu Li, Richong Zhang, Tongwen Huang, Xiangxiang Chu, Yong Wang, Zixuan Huang","submitted_at":"2024-11-21T12:16:16Z","abstract_excerpt":"Large Multimodal Models (LMMs) demonstrate impressive capabilities. However, current benchmarks predominantly focus on image comprehension in specific domains, and these benchmarks are labor-intensive to construct. Moreover, their answers tend to be brief, making it difficult to assess the ability of LMMs to generate detailed descriptions of images. To address these limitations, we propose the MMGenBench-Pipeline, a straightforward and fully automated evaluation pipeline. This involves generating textual descriptions from input images, using these descriptions to create auxiliary images via te"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.14062","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-21T12:16:16Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"fd0501e9cef1956d2a3dc7502276f69e663e2dc5ef856a2c97bd7466248f6aee","abstract_canon_sha256":"3dcc544b813b7bcde9b1d8cd678b6aefeb62cb8ecbc7db89b169e1fffe3ab089"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:41.568737Z","signature_b64":"05sX9cptWWGzEV1CPNrRp3HFnHVNYFV0rBjt+R7JtivYGT3U5Os+KgXKn5tvkFBTsAJ76Z0/o1s7nN8COxJUCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80b23b42a26b712e57d04889b5fe1da08ac9ff00aef8ff179d435e145cda1613","last_reissued_at":"2026-07-05T10:26:41.567856Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:41.567856Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hailang Huang, HuaQiu Li, Richong Zhang, Tongwen Huang, Xiangxiang Chu, Yong Wang, Zixuan Huang","submitted_at":"2024-11-21T12:16:16Z","abstract_excerpt":"Large Multimodal Models (LMMs) demonstrate impressive capabilities. However, current benchmarks predominantly focus on image comprehension in specific domains, and these benchmarks are labor-intensive to construct. Moreover, their answers tend to be brief, making it difficult to assess the ability of LMMs to generate detailed descriptions of images. To address these limitations, we propose the MMGenBench-Pipeline, a straightforward and fully automated evaluation pipeline. This involves generating textual descriptions from input images, using these descriptions to create auxiliary images via te"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14062","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14062/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.14062","created_at":"2026-07-05T10:26:41.567978+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.14062v2","created_at":"2026-07-05T10:26:41.567978+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14062","created_at":"2026-07-05T10:26:41.567978+00:00"},{"alias_kind":"pith_short_12","alias_value":"QCZDWQVCNNYS","created_at":"2026-07-05T10:26:41.567978+00:00"},{"alias_kind":"pith_short_16","alias_value":"QCZDWQVCNNYS4V6Q","created_at":"2026-07-05T10:26:41.567978+00:00"},{"alias_kind":"pith_short_8","alias_value":"QCZDWQVC","created_at":"2026-07-05T10:26:41.567978+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.08039","citing_title":"Towards Evaluating Robustness of Prompt Adherence in Text to Image Models","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC","json":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC.json","graph_json":"https://pith.science/api/pith-number/QCZDWQVCNNYS4V6QJCE3L7Q5UC/graph.json","events_json":"https://pith.science/api/pith-number/QCZDWQVCNNYS4V6QJCE3L7Q5UC/events.json","paper":"https://pith.science/paper/QCZDWQVC"},"agent_actions":{"view_html":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC","download_json":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC.json","view_paper":"https://pith.science/paper/QCZDWQVC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.14062&json=true","fetch_graph":"https://pith.science/api/pith-number/QCZDWQVCNNYS4V6QJCE3L7Q5UC/graph.json","fetch_events":"https://pith.science/api/pith-number/QCZDWQVCNNYS4V6QJCE3L7Q5UC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC/action/storage_attestation","attest_author":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC/action/author_attestation","sign_citation":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC/action/citation_signature","submit_replication":"https://pith.science/pith/QCZDWQVCNNYS4V6QJCE3L7Q5UC/action/replication_record"}},"created_at":"2026-07-05T10:26:41.567978+00:00","updated_at":"2026-07-05T10:26:41.567978+00:00"}