{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:E5AVXN6TXG4EAF5FVTU5UHCLMS","short_pith_number":"pith:E5AVXN6T","schema_version":"1.0","canonical_sha256":"27415bb7d3b9b84017a5ace9da1c4b64b61d83514bee6864f2c9db63bfccb137","source":{"kind":"arxiv","id":"2607.15778","version":1},"attestation_state":"computed","paper":{"title":"Modularized Dynamic-Granularity Video LLM for Multi-Event Long Video Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Wei Feng, Wenwu Zhu, Xin Wang, Yu-Wei Zhan, Yuwei Zhou","submitted_at":"2026-07-17T09:21:52Z","abstract_excerpt":"Video Large Language Models (Video LLMs) have made significant advancements in various video understanding tasks. However, long-video scenarios remain challenging due to the tension between limited visual token budgets and the need to capture multiple key events. Existing approaches typically process long videos in two stages, i.e., i) select keyframes and ii) perform detailed perception, which exhibit limitations: they lack a modular mechanism for adaptive capacity allocation and self-correction, resulting in unreliable modeling. To tackle these challenges, we propose MoD-VLLM, a novel Modula"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.15778","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2026-07-17T09:21:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a6b665801325c5302d040dc01775b29978d625dde820897d8f90b6e716ddc229","abstract_canon_sha256":"0de1d097e61bdee07f136f6f54432af28cce7cb749ac0ed8b2027bb1df0b9a62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-20T01:19:09.618621Z","signature_b64":"bi/bbClB/pnzd7Ym2fwYi/y9e+aF8ZdHpvXtvZ51DtaXnjqz9d4UajBCF2480Ih9Wcljxli+0nQC9cytpCIuBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27415bb7d3b9b84017a5ace9da1c4b64b61d83514bee6864f2c9db63bfccb137","last_reissued_at":"2026-07-20T01:19:09.617705Z","signature_status":"signed_v1","first_computed_at":"2026-07-20T01:19:09.617705Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Modularized Dynamic-Granularity Video LLM for Multi-Event Long Video Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Wei Feng, Wenwu Zhu, Xin Wang, Yu-Wei Zhan, Yuwei Zhou","submitted_at":"2026-07-17T09:21:52Z","abstract_excerpt":"Video Large Language Models (Video LLMs) have made significant advancements in various video understanding tasks. However, long-video scenarios remain challenging due to the tension between limited visual token budgets and the need to capture multiple key events. Existing approaches typically process long videos in two stages, i.e., i) select keyframes and ii) perform detailed perception, which exhibit limitations: they lack a modular mechanism for adaptive capacity allocation and self-correction, resulting in unreliable modeling. To tackle these challenges, we propose MoD-VLLM, a novel Modula"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.15778","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.15778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.15778","created_at":"2026-07-20T01:19:09.618186+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.15778v1","created_at":"2026-07-20T01:19:09.618186+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.15778","created_at":"2026-07-20T01:19:09.618186+00:00"},{"alias_kind":"pith_short_12","alias_value":"E5AVXN6TXG4E","created_at":"2026-07-20T01:19:09.618186+00:00"},{"alias_kind":"pith_short_16","alias_value":"E5AVXN6TXG4EAF5F","created_at":"2026-07-20T01:19:09.618186+00:00"},{"alias_kind":"pith_short_8","alias_value":"E5AVXN6T","created_at":"2026-07-20T01:19:09.618186+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS","json":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS.json","graph_json":"https://pith.science/api/pith-number/E5AVXN6TXG4EAF5FVTU5UHCLMS/graph.json","events_json":"https://pith.science/api/pith-number/E5AVXN6TXG4EAF5FVTU5UHCLMS/events.json","paper":"https://pith.science/paper/E5AVXN6T"},"agent_actions":{"view_html":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS","download_json":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS.json","view_paper":"https://pith.science/paper/E5AVXN6T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.15778&json=true","fetch_graph":"https://pith.science/api/pith-number/E5AVXN6TXG4EAF5FVTU5UHCLMS/graph.json","fetch_events":"https://pith.science/api/pith-number/E5AVXN6TXG4EAF5FVTU5UHCLMS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS/action/storage_attestation","attest_author":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS/action/author_attestation","sign_citation":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS/action/citation_signature","submit_replication":"https://pith.science/pith/E5AVXN6TXG4EAF5FVTU5UHCLMS/action/replication_record"}},"created_at":"2026-07-20T01:19:09.618186+00:00","updated_at":"2026-07-20T01:19:09.618186+00:00"}