{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XDYU3LP7JLZ6EPLJ6AYVU44BI7","short_pith_number":"pith:XDYU3LP7","schema_version":"1.0","canonical_sha256":"b8f14dadff4af3e23d69f0315a738147e50170f1c741745d0d57a833978f981a","source":{"kind":"arxiv","id":"2404.03413","version":1},"attestation_state":"computed","paper":{"title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deyao Zhu, Eslam Abdelrahman, Essam Sleiman, Jian Ding, Kirolos Ataallah, Mohamed Elhoseiny, Xiaoqian Shen","submitted_at":"2024-04-04T12:46:01Z","abstract_excerpt":"This paper introduces MiniGPT4-Video, a multimodal Large Language Model (LLM) designed specifically for video understanding. The model is capable of processing both temporal visual and textual data, making it adept at understanding the complexities of videos. Building upon the success of MiniGPT-v2, which excelled in translating visual features into the LLM space for single images and achieved impressive results on various image-text benchmarks, this paper extends the model's capabilities to process a sequence of frames, enabling it to comprehend videos. MiniGPT4-video does not only consider v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03413","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-04T12:46:01Z","cross_cats_sorted":[],"title_canon_sha256":"1b31a107d5f47599dfa70006f4210edadff6ec700c5cd018de04f1638d386891","abstract_canon_sha256":"224fef405f4edd4f06cd0d34f0b4b2692bdbd8632e3120a232873ece62b7e880"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:25.463120Z","signature_b64":"uGgvgxp8e7d+HBzVeWDODNUsdmEqKpm5y4DiccUOleKbVC7bDPpCVv6ShiFTgJr6/596Bb9YVqG2fFoohLX6BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8f14dadff4af3e23d69f0315a738147e50170f1c741745d0d57a833978f981a","last_reissued_at":"2026-07-05T08:04:25.462668Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:25.462668Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deyao Zhu, Eslam Abdelrahman, Essam Sleiman, Jian Ding, Kirolos Ataallah, Mohamed Elhoseiny, Xiaoqian Shen","submitted_at":"2024-04-04T12:46:01Z","abstract_excerpt":"This paper introduces MiniGPT4-Video, a multimodal Large Language Model (LLM) designed specifically for video understanding. The model is capable of processing both temporal visual and textual data, making it adept at understanding the complexities of videos. Building upon the success of MiniGPT-v2, which excelled in translating visual features into the LLM space for single images and achieved impressive results on various image-text benchmarks, this paper extends the model's capabilities to process a sequence of frames, enabling it to comprehend videos. MiniGPT4-video does not only consider v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03413","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03413/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03413","created_at":"2026-07-05T08:04:25.462726+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03413v1","created_at":"2026-07-05T08:04:25.462726+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03413","created_at":"2026-07-05T08:04:25.462726+00:00"},{"alias_kind":"pith_short_12","alias_value":"XDYU3LP7JLZ6","created_at":"2026-07-05T08:04:25.462726+00:00"},{"alias_kind":"pith_short_16","alias_value":"XDYU3LP7JLZ6EPLJ","created_at":"2026-07-05T08:04:25.462726+00:00"},{"alias_kind":"pith_short_8","alias_value":"XDYU3LP7","created_at":"2026-07-05T08:04:25.462726+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17798","citing_title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06991","citing_title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30288","citing_title":"VisReflect: Latent Visual Reflection for Fine-Grained Perception in Long Visual Context","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05067","citing_title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21998","citing_title":"Can Multi-Modal LLMs Provide Live Step-by-Step Task Guidance?","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17434","citing_title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04264","citing_title":"MLVU: Benchmarking Multi-task Long Video Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07476","citing_title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13106","citing_title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05079","citing_title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7","json":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7.json","graph_json":"https://pith.science/api/pith-number/XDYU3LP7JLZ6EPLJ6AYVU44BI7/graph.json","events_json":"https://pith.science/api/pith-number/XDYU3LP7JLZ6EPLJ6AYVU44BI7/events.json","paper":"https://pith.science/paper/XDYU3LP7"},"agent_actions":{"view_html":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7","download_json":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7.json","view_paper":"https://pith.science/paper/XDYU3LP7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03413&json=true","fetch_graph":"https://pith.science/api/pith-number/XDYU3LP7JLZ6EPLJ6AYVU44BI7/graph.json","fetch_events":"https://pith.science/api/pith-number/XDYU3LP7JLZ6EPLJ6AYVU44BI7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7/action/storage_attestation","attest_author":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7/action/author_attestation","sign_citation":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7/action/citation_signature","submit_replication":"https://pith.science/pith/XDYU3LP7JLZ6EPLJ6AYVU44BI7/action/replication_record"}},"created_at":"2026-07-05T08:04:25.462726+00:00","updated_at":"2026-07-05T08:04:25.462726+00:00"}