{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BJUHOX7HOXB6MCNPHDOCTUDWOK","short_pith_number":"pith:BJUHOX7H","schema_version":"1.0","canonical_sha256":"0a68775fe775c3e609af38dc29d07672a8566530cc824020c73c0150c2974843","source":{"kind":"arxiv","id":"2501.15147","version":2},"attestation_state":"computed","paper":{"title":"A Causality-aware Paradigm for Evaluating Creativity of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.AI","authors_text":"Liang Lin, Marinka Zitnik, Pan Zhou, Shanghua Gao, Shanshan Zhong, Zhongzhan Huang","submitted_at":"2025-01-25T09:11:15Z","abstract_excerpt":"Recently, numerous benchmarks have been developed to evaluate the logical reasoning abilities of large language models (LLMs). However, assessing the equally important creative capabilities of LLMs is challenging due to the subjective, diverse, and data-scarce nature of creativity, especially in multimodal scenarios. In this paper, we consider the comprehensive pipeline for evaluating the creativity of multimodal LLMs, with a focus on suitable evaluation platforms and methodologies. First, we find the Oogiri game, a creativity-driven task requiring humor, associative thinking, and the ability "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.15147","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-01-25T09:11:15Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"b141998cc075447776bd7544391e9ecb3c6210b9cd3fcb591c5c5d3bbed077c5","abstract_canon_sha256":"1e009a2d742e027b0b061e09abb4b48c88026476ad944263b6dbcbe8895dd2d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:48.734442Z","signature_b64":"xtz0QKSn9DuqdN9KlgpiMG9LAjaAZIrm0Aq10Gh8HZWqGY4X7FrmxuOIyC6zFfyzYm0Ybqe8JU2KFUb2yMhwDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a68775fe775c3e609af38dc29d07672a8566530cc824020c73c0150c2974843","last_reissued_at":"2026-07-05T10:18:48.731907Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:48.731907Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Causality-aware Paradigm for Evaluating Creativity of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.AI","authors_text":"Liang Lin, Marinka Zitnik, Pan Zhou, Shanghua Gao, Shanshan Zhong, Zhongzhan Huang","submitted_at":"2025-01-25T09:11:15Z","abstract_excerpt":"Recently, numerous benchmarks have been developed to evaluate the logical reasoning abilities of large language models (LLMs). However, assessing the equally important creative capabilities of LLMs is challenging due to the subjective, diverse, and data-scarce nature of creativity, especially in multimodal scenarios. In this paper, we consider the comprehensive pipeline for evaluating the creativity of multimodal LLMs, with a focus on suitable evaluation platforms and methodologies. First, we find the Oogiri game, a creativity-driven task requiring humor, associative thinking, and the ability "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.15147","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.15147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.15147","created_at":"2026-07-05T10:18:48.731984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.15147v2","created_at":"2026-07-05T10:18:48.731984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.15147","created_at":"2026-07-05T10:18:48.731984+00:00"},{"alias_kind":"pith_short_12","alias_value":"BJUHOX7HOXB6","created_at":"2026-07-05T10:18:48.731984+00:00"},{"alias_kind":"pith_short_16","alias_value":"BJUHOX7HOXB6MCNP","created_at":"2026-07-05T10:18:48.731984+00:00"},{"alias_kind":"pith_short_8","alias_value":"BJUHOX7H","created_at":"2026-07-05T10:18:48.731984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK","json":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK.json","graph_json":"https://pith.science/api/pith-number/BJUHOX7HOXB6MCNPHDOCTUDWOK/graph.json","events_json":"https://pith.science/api/pith-number/BJUHOX7HOXB6MCNPHDOCTUDWOK/events.json","paper":"https://pith.science/paper/BJUHOX7H"},"agent_actions":{"view_html":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK","download_json":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK.json","view_paper":"https://pith.science/paper/BJUHOX7H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.15147&json=true","fetch_graph":"https://pith.science/api/pith-number/BJUHOX7HOXB6MCNPHDOCTUDWOK/graph.json","fetch_events":"https://pith.science/api/pith-number/BJUHOX7HOXB6MCNPHDOCTUDWOK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK/action/storage_attestation","attest_author":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK/action/author_attestation","sign_citation":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK/action/citation_signature","submit_replication":"https://pith.science/pith/BJUHOX7HOXB6MCNPHDOCTUDWOK/action/replication_record"}},"created_at":"2026-07-05T10:18:48.731984+00:00","updated_at":"2026-07-05T10:18:48.731984+00:00"}