{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KQB7S4RDRKWR4EY4PDPR3DCQJD","short_pith_number":"pith:KQB7S4RD","schema_version":"1.0","canonical_sha256":"5403f972238aad1e131c78df1d8c5048fc2bbeffbaff6ff0b5c29d205ece3200","source":{"kind":"arxiv","id":"2405.03272","version":1},"attestation_state":"computed","paper":{"title":"WorldQA: Multimodal World Knowledge in Videos through Long-Chain Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Li, Christopher Arif Setiadharma, Fanyi Pu, Jingkang Yang, Kaichen Zhang, Yuanhan Zhang, Ziwei Liu","submitted_at":"2024-05-06T08:42:34Z","abstract_excerpt":"Multimodal information, together with our knowledge, help us to understand the complex and dynamic world. Large language models (LLM) and large multimodal models (LMM), however, still struggle to emulate this capability. In this paper, we present WorldQA, a video understanding dataset designed to push the boundaries of multimodal world models with three appealing properties: (1) Multimodal Inputs: The dataset comprises 1007 question-answer pairs and 303 videos, necessitating the analysis of both auditory and visual data for successful interpretation. (2) World Knowledge: We identify five essen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.03272","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-06T08:42:34Z","cross_cats_sorted":[],"title_canon_sha256":"8c61891552dd203f73ccce24e374a35250af207d1af54d32c215f1a5075398b5","abstract_canon_sha256":"5e33a23857d93cc24ddf343df47801e2149123468b468a35ba652cc24ecdbc83"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:16:00.885836Z","signature_b64":"o31HHG81vEnkqOPhheuKP4Tuw01ekS6D+ugQWfMl/1BJ84+fRHNHIQ03FSF2nS50qBsdoZGXEeoEAx75U+WOAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5403f972238aad1e131c78df1d8c5048fc2bbeffbaff6ff0b5c29d205ece3200","last_reissued_at":"2026-07-05T08:16:00.885360Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:16:00.885360Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WorldQA: Multimodal World Knowledge in Videos through Long-Chain Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Li, Christopher Arif Setiadharma, Fanyi Pu, Jingkang Yang, Kaichen Zhang, Yuanhan Zhang, Ziwei Liu","submitted_at":"2024-05-06T08:42:34Z","abstract_excerpt":"Multimodal information, together with our knowledge, help us to understand the complex and dynamic world. Large language models (LLM) and large multimodal models (LMM), however, still struggle to emulate this capability. In this paper, we present WorldQA, a video understanding dataset designed to push the boundaries of multimodal world models with three appealing properties: (1) Multimodal Inputs: The dataset comprises 1007 question-answer pairs and 303 videos, necessitating the analysis of both auditory and visual data for successful interpretation. (2) World Knowledge: We identify five essen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.03272","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.03272/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.03272","created_at":"2026-07-05T08:16:00.885419+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.03272v1","created_at":"2026-07-05T08:16:00.885419+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.03272","created_at":"2026-07-05T08:16:00.885419+00:00"},{"alias_kind":"pith_short_12","alias_value":"KQB7S4RDRKWR","created_at":"2026-07-05T08:16:00.885419+00:00"},{"alias_kind":"pith_short_16","alias_value":"KQB7S4RDRKWR4EY4","created_at":"2026-07-05T08:16:00.885419+00:00"},{"alias_kind":"pith_short_8","alias_value":"KQB7S4RD","created_at":"2026-07-05T08:16:00.885419+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07643","citing_title":"AVI-Bench: Toward Human-like Audio-Visual Intelligence of Omni-MLLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13826","citing_title":"Video-MMMU: Evaluating Knowledge Acquisition from Multi-Discipline Professional Videos","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD","json":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD.json","graph_json":"https://pith.science/api/pith-number/KQB7S4RDRKWR4EY4PDPR3DCQJD/graph.json","events_json":"https://pith.science/api/pith-number/KQB7S4RDRKWR4EY4PDPR3DCQJD/events.json","paper":"https://pith.science/paper/KQB7S4RD"},"agent_actions":{"view_html":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD","download_json":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD.json","view_paper":"https://pith.science/paper/KQB7S4RD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.03272&json=true","fetch_graph":"https://pith.science/api/pith-number/KQB7S4RDRKWR4EY4PDPR3DCQJD/graph.json","fetch_events":"https://pith.science/api/pith-number/KQB7S4RDRKWR4EY4PDPR3DCQJD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD/action/storage_attestation","attest_author":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD/action/author_attestation","sign_citation":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD/action/citation_signature","submit_replication":"https://pith.science/pith/KQB7S4RDRKWR4EY4PDPR3DCQJD/action/replication_record"}},"created_at":"2026-07-05T08:16:00.885419+00:00","updated_at":"2026-07-05T08:16:00.885419+00:00"}