{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7OKOJUDHQONAQOERBA4SAQCU4F","short_pith_number":"pith:7OKOJUDH","schema_version":"1.0","canonical_sha256":"fb94e4d067839a0838910839204054e1692702c96fc0766b201d32e677baccff","source":{"kind":"arxiv","id":"2403.18406","version":1},"attestation_state":"computed","paper":{"title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Changin Choi, Wonjong Rhee, Wonkyun Kim, Wonseok Lee","submitted_at":"2024-03-27T09:48:23Z","abstract_excerpt":"Stimulated by the sophisticated reasoning capabilities of recent Large Language Models (LLMs), a variety of strategies for bridging video modality have been devised. A prominent strategy involves Video Language Models (VideoLMs), which train a learnable interface with video data to connect advanced vision encoders with LLMs. Recently, an alternative strategy has surfaced, employing readily available foundation models, such as VideoLMs and LLMs, across multiple stages for modality bridging. In this study, we introduce a simple yet novel strategy where only a single Vision Language Model (VLM) i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.18406","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-27T09:48:23Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"d4ef7a152ff7799423b65730984daef6268ed4c2c62bf5887eaeb727f9cfbfc1","abstract_canon_sha256":"9b72fd919953e208271e4037fc9ee1cbac48a8130293741321f3227b03bf4acc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:22.234170Z","signature_b64":"MsW65xAaUd8gZo+NZ+aCtHw+Kdk2i+wvWe/1GjBAFnE7E/orTcO/w67h5gkng/sgMxdjsbZpYmx2IkrwAq7kDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb94e4d067839a0838910839204054e1692702c96fc0766b201d32e677baccff","last_reissued_at":"2026-07-05T08:01:22.233550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:22.233550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Changin Choi, Wonjong Rhee, Wonkyun Kim, Wonseok Lee","submitted_at":"2024-03-27T09:48:23Z","abstract_excerpt":"Stimulated by the sophisticated reasoning capabilities of recent Large Language Models (LLMs), a variety of strategies for bridging video modality have been devised. A prominent strategy involves Video Language Models (VideoLMs), which train a learnable interface with video data to connect advanced vision encoders with LLMs. Recently, an alternative strategy has surfaced, employing readily available foundation models, such as VideoLMs and LLMs, across multiple stages for modality bridging. In this study, we introduce a simple yet novel strategy where only a single Vision Language Model (VLM) i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.18406","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.18406/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.18406","created_at":"2026-07-05T08:01:22.233637+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.18406v1","created_at":"2026-07-05T08:01:22.233637+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.18406","created_at":"2026-07-05T08:01:22.233637+00:00"},{"alias_kind":"pith_short_12","alias_value":"7OKOJUDHQONA","created_at":"2026-07-05T08:01:22.233637+00:00"},{"alias_kind":"pith_short_16","alias_value":"7OKOJUDHQONAQOER","created_at":"2026-07-05T08:01:22.233637+00:00"},{"alias_kind":"pith_short_8","alias_value":"7OKOJUDH","created_at":"2026-07-05T08:01:22.233637+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.02955","citing_title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22078","citing_title":"Enhancing Visual Token Representations for Video Large Language Models via Training-Free Spatial-Temporal Pooling and Gridding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16994","citing_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02891","citing_title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01662","citing_title":"Video Active Perception: Effective Inference-Time Long-Form Video Understanding with Vision-Language Models","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F","json":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F.json","graph_json":"https://pith.science/api/pith-number/7OKOJUDHQONAQOERBA4SAQCU4F/graph.json","events_json":"https://pith.science/api/pith-number/7OKOJUDHQONAQOERBA4SAQCU4F/events.json","paper":"https://pith.science/paper/7OKOJUDH"},"agent_actions":{"view_html":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F","download_json":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F.json","view_paper":"https://pith.science/paper/7OKOJUDH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.18406&json=true","fetch_graph":"https://pith.science/api/pith-number/7OKOJUDHQONAQOERBA4SAQCU4F/graph.json","fetch_events":"https://pith.science/api/pith-number/7OKOJUDHQONAQOERBA4SAQCU4F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F/action/storage_attestation","attest_author":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F/action/author_attestation","sign_citation":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F/action/citation_signature","submit_replication":"https://pith.science/pith/7OKOJUDHQONAQOERBA4SAQCU4F/action/replication_record"}},"created_at":"2026-07-05T08:01:22.233637+00:00","updated_at":"2026-07-05T08:01:22.233637+00:00"}