{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QBNO4ZP6RHKZOOFAKHQOXZ6FKS","short_pith_number":"pith:QBNO4ZP6","schema_version":"1.0","canonical_sha256":"805aee65fe89d59738a051e0ebe7c554944b02b252c010845a8b11f063bb4674","source":{"kind":"arxiv","id":"2310.11513","version":1},"attestation_state":"computed","paper":{"title":"GenEval: An Object-Focused Framework for Evaluating Text-to-Image Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Dhruba Ghosh, Hanna Hajishirzi, Ludwig Schmidt","submitted_at":"2023-10-17T18:20:03Z","abstract_excerpt":"Recent breakthroughs in diffusion models, multimodal pretraining, and efficient finetuning have led to an explosion of text-to-image generative models. Given human evaluation is expensive and difficult to scale, automated methods are critical for evaluating the increasingly large number of new models. However, most current automated evaluation metrics like FID or CLIPScore only offer a holistic measure of image quality or image-text alignment, and are unsuited for fine-grained or instance-level analysis. In this paper, we introduce GenEval, an object-focused framework to evaluate compositional"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.11513","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-17T18:20:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"70dbf7843e5a142c8c6546c8fad0f3fe1ad130ec7997a7dee2b25dd7806a3bfe","abstract_canon_sha256":"c164cd140a5c4ccf308a16b52ea0fa29223b6dfdba8cded4cb2f3b3bb57f98f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:08.419537Z","signature_b64":"22tz4FrkbK+q0OGaobg5VzjUGM0z865vVuAca9/tLl4dPKepRPtVugvJPgnEcTKJqa6kP91gl+INtlQ87GDoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"805aee65fe89d59738a051e0ebe7c554944b02b252c010845a8b11f063bb4674","last_reissued_at":"2026-07-05T07:02:08.419070Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:08.419070Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GenEval: An Object-Focused Framework for Evaluating Text-to-Image Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Dhruba Ghosh, Hanna Hajishirzi, Ludwig Schmidt","submitted_at":"2023-10-17T18:20:03Z","abstract_excerpt":"Recent breakthroughs in diffusion models, multimodal pretraining, and efficient finetuning have led to an explosion of text-to-image generative models. Given human evaluation is expensive and difficult to scale, automated methods are critical for evaluating the increasingly large number of new models. However, most current automated evaluation metrics like FID or CLIPScore only offer a holistic measure of image quality or image-text alignment, and are unsuited for fine-grained or instance-level analysis. In this paper, we introduce GenEval, an object-focused framework to evaluate compositional"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.11513","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.11513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.11513","created_at":"2026-07-05T07:02:08.419125+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.11513v1","created_at":"2026-07-05T07:02:08.419125+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.11513","created_at":"2026-07-05T07:02:08.419125+00:00"},{"alias_kind":"pith_short_12","alias_value":"QBNO4ZP6RHKZ","created_at":"2026-07-05T07:02:08.419125+00:00"},{"alias_kind":"pith_short_16","alias_value":"QBNO4ZP6RHKZOOFA","created_at":"2026-07-05T07:02:08.419125+00:00"},{"alias_kind":"pith_short_8","alias_value":"QBNO4ZP6","created_at":"2026-07-05T07:02:08.419125+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26907","citing_title":"Qwen-Image-Agent: Bridging the Context Gap in Real-World Image Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00402","citing_title":"The Illusion of High Utility in Safety Alignment of Text-to-Image Diffusion Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26907","citing_title":"Qwen-Image-Agent: Bridging the Context Gap in Real-World Image Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27771","citing_title":"NormGuard: Reward-Preserving Norm Constraints in Flow-Matching Reinforcement Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21195","citing_title":"RankE: End-to-End Post-Training for Discrete Text-to-Image Generation with Decoder Co-Evolution","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14876","citing_title":"Unlocking Complex Visual Generation via Closed-Loop Verified Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21912","citing_title":"Discrete Guidance Matching: Exact Guidance for Discrete Flow Matching","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06165","citing_title":"Reflective Flow Sampling Enhancement","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03206","citing_title":"Scaling Rectified Flow Transformers for High-Resolution Image Synthesis","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25072","citing_title":"Beyond Accuracy: Benchmarking Cross-Task Consistency in Unified Multimodal Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23540","citing_title":"Oracle Noise: Faster Semantic Spherical Alignment for Interpretable Latent Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04746","citing_title":"Think in Strokes, Not Pixels: Process-Driven Image Generation via Interleaved Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13540","citing_title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS","json":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS.json","graph_json":"https://pith.science/api/pith-number/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/graph.json","events_json":"https://pith.science/api/pith-number/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/events.json","paper":"https://pith.science/paper/QBNO4ZP6"},"agent_actions":{"view_html":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS","download_json":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS.json","view_paper":"https://pith.science/paper/QBNO4ZP6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.11513&json=true","fetch_graph":"https://pith.science/api/pith-number/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/graph.json","fetch_events":"https://pith.science/api/pith-number/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/action/storage_attestation","attest_author":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/action/author_attestation","sign_citation":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/action/citation_signature","submit_replication":"https://pith.science/pith/QBNO4ZP6RHKZOOFAKHQOXZ6FKS/action/replication_record"}},"created_at":"2026-07-05T07:02:08.419125+00:00","updated_at":"2026-07-05T07:02:08.419125+00:00"}