{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S6AVLRVM7RG7UUW6YJH556D7KP","short_pith_number":"pith:S6AVLRVM","schema_version":"1.0","canonical_sha256":"978155c6acfc4dfa52dec24fdef87f53e6d7e3ee12c4c98c2ea49a9dccb76373","source":{"kind":"arxiv","id":"2406.14643","version":3},"attestation_state":"computed","paper":{"title":"Holistic Evaluation for Interleaved Text-and-Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiaxin Zhang, Joy Rimchala, Lifu Huang, Minqian Liu, Trevor Ashby, Zhiyang Xu, Zihao Lin","submitted_at":"2024-06-20T18:07:19Z","abstract_excerpt":"Interleaved text-and-image generation has been an intriguing research direction, where the models are required to generate both images and text pieces in an arbitrary order. Despite the emerging advancements in interleaved generation, the progress in its evaluation still significantly lags behind. Existing evaluation benchmarks do not support arbitrarily interleaved images and text for both inputs and outputs, and they only cover a limited number of domains and use cases. Also, current works predominantly use similarity-based metrics which fall short in assessing the quality in open-ended scen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.14643","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-20T18:07:19Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"beeb9057c6f2a98c4c09dfa5f89ef4dec7764420358b05752d308abd40f72d52","abstract_canon_sha256":"67e29b87cb1d2b94aac1a6a59db4dd3d01e4c3c0bbf34721c6b26a642dda415a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:44.438243Z","signature_b64":"ZhcOqPgBi26kcw9fICgis80LKYrNpUDnPQRYILb1S8vI0CW6CDgbleFS2e63U2+t9qFmDsllBCD0r72KPF4tDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"978155c6acfc4dfa52dec24fdef87f53e6d7e3ee12c4c98c2ea49a9dccb76373","last_reissued_at":"2026-07-05T09:17:44.437776Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:44.437776Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Holistic Evaluation for Interleaved Text-and-Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiaxin Zhang, Joy Rimchala, Lifu Huang, Minqian Liu, Trevor Ashby, Zhiyang Xu, Zihao Lin","submitted_at":"2024-06-20T18:07:19Z","abstract_excerpt":"Interleaved text-and-image generation has been an intriguing research direction, where the models are required to generate both images and text pieces in an arbitrary order. Despite the emerging advancements in interleaved generation, the progress in its evaluation still significantly lags behind. Existing evaluation benchmarks do not support arbitrarily interleaved images and text for both inputs and outputs, and they only cover a limited number of domains and use cases. Also, current works predominantly use similarity-based metrics which fall short in assessing the quality in open-ended scen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.14643","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.14643/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.14643","created_at":"2026-07-05T09:17:44.437834+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.14643v3","created_at":"2026-07-05T09:17:44.437834+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.14643","created_at":"2026-07-05T09:17:44.437834+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6AVLRVM7RG7","created_at":"2026-07-05T09:17:44.437834+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6AVLRVM7RG7UUW6","created_at":"2026-07-05T09:17:44.437834+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6AVLRVM","created_at":"2026-07-05T09:17:44.437834+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17296","citing_title":"Pareto LoRA: Mitigating Modality Imbalance in Unified Multimodal Models via Pareto-Optimal Gradient Integration","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP","json":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP.json","graph_json":"https://pith.science/api/pith-number/S6AVLRVM7RG7UUW6YJH556D7KP/graph.json","events_json":"https://pith.science/api/pith-number/S6AVLRVM7RG7UUW6YJH556D7KP/events.json","paper":"https://pith.science/paper/S6AVLRVM"},"agent_actions":{"view_html":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP","download_json":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP.json","view_paper":"https://pith.science/paper/S6AVLRVM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.14643&json=true","fetch_graph":"https://pith.science/api/pith-number/S6AVLRVM7RG7UUW6YJH556D7KP/graph.json","fetch_events":"https://pith.science/api/pith-number/S6AVLRVM7RG7UUW6YJH556D7KP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP/action/storage_attestation","attest_author":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP/action/author_attestation","sign_citation":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP/action/citation_signature","submit_replication":"https://pith.science/pith/S6AVLRVM7RG7UUW6YJH556D7KP/action/replication_record"}},"created_at":"2026-07-05T09:17:44.437834+00:00","updated_at":"2026-07-05T09:17:44.437834+00:00"}