{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OGH5FOVCIY4AXORFO6W23XPAAD","short_pith_number":"pith:OGH5FOVC","schema_version":"1.0","canonical_sha256":"718fd2baa246380bba2577adaddde000c47a316d4d821a81ad2122463402ec74","source":{"kind":"arxiv","id":"2508.21044","version":1},"attestation_state":"computed","paper":{"title":"MMG-Vid: Maximizing Marginal Gains at Segment-level and Token-level for Efficient Video LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junpeng Ma, Jun Song, Ming Lu, Qiang Zhou, Qizhe Zhang, Shanghang Zhang, Zhibin Wang","submitted_at":"2025-08-28T17:50:03Z","abstract_excerpt":"Video Large Language Models (VLLMs) excel in video understanding, but their excessive visual tokens pose a significant computational challenge for real-world applications. Current methods aim to enhance inference efficiency by visual token pruning. However, they do not consider the dynamic characteristics and temporal dependencies of video frames, as they perceive video understanding as a multi-frame task. To address these challenges, we propose MMG-Vid, a novel training-free visual token pruning framework that removes redundancy by Maximizing Marginal Gains at both segment-level and token-lev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.21044","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-28T17:50:03Z","cross_cats_sorted":[],"title_canon_sha256":"64dc3f71031197822cd9045933135b469d2f8641796236da95c5640c572bc513","abstract_canon_sha256":"abbad899e637600cb54b93533faee15eb25a5759bc0d23bc74997b2b5a65a298"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:15.771022Z","signature_b64":"88cARpqVnFpFpWacSvp+Yzd3GxwD9eBS2ofrRkyUgCVlZEcFAcJ/2PslbKCcbodEGmOnus2yRvH1pn8MFG7fDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"718fd2baa246380bba2577adaddde000c47a316d4d821a81ad2122463402ec74","last_reissued_at":"2026-07-05T12:01:15.770525Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:15.770525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMG-Vid: Maximizing Marginal Gains at Segment-level and Token-level for Efficient Video LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junpeng Ma, Jun Song, Ming Lu, Qiang Zhou, Qizhe Zhang, Shanghang Zhang, Zhibin Wang","submitted_at":"2025-08-28T17:50:03Z","abstract_excerpt":"Video Large Language Models (VLLMs) excel in video understanding, but their excessive visual tokens pose a significant computational challenge for real-world applications. Current methods aim to enhance inference efficiency by visual token pruning. However, they do not consider the dynamic characteristics and temporal dependencies of video frames, as they perceive video understanding as a multi-frame task. To address these challenges, we propose MMG-Vid, a novel training-free visual token pruning framework that removes redundancy by Maximizing Marginal Gains at both segment-level and token-lev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21044","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.21044/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.21044","created_at":"2026-07-05T12:01:15.770585+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.21044v1","created_at":"2026-07-05T12:01:15.770585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21044","created_at":"2026-07-05T12:01:15.770585+00:00"},{"alias_kind":"pith_short_12","alias_value":"OGH5FOVCIY4A","created_at":"2026-07-05T12:01:15.770585+00:00"},{"alias_kind":"pith_short_16","alias_value":"OGH5FOVCIY4AXORF","created_at":"2026-07-05T12:01:15.770585+00:00"},{"alias_kind":"pith_short_8","alias_value":"OGH5FOVC","created_at":"2026-07-05T12:01:15.770585+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04055","citing_title":"DINO-VO: Learning Where to Focus for Enhanced State Estimation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20937","citing_title":"Sink-Token-Aware Pruning for Fine-Grained Video Understanding in Efficient Video LLMs","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD","json":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD.json","graph_json":"https://pith.science/api/pith-number/OGH5FOVCIY4AXORFO6W23XPAAD/graph.json","events_json":"https://pith.science/api/pith-number/OGH5FOVCIY4AXORFO6W23XPAAD/events.json","paper":"https://pith.science/paper/OGH5FOVC"},"agent_actions":{"view_html":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD","download_json":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD.json","view_paper":"https://pith.science/paper/OGH5FOVC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.21044&json=true","fetch_graph":"https://pith.science/api/pith-number/OGH5FOVCIY4AXORFO6W23XPAAD/graph.json","fetch_events":"https://pith.science/api/pith-number/OGH5FOVCIY4AXORFO6W23XPAAD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD/action/storage_attestation","attest_author":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD/action/author_attestation","sign_citation":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD/action/citation_signature","submit_replication":"https://pith.science/pith/OGH5FOVCIY4AXORFO6W23XPAAD/action/replication_record"}},"created_at":"2026-07-05T12:01:15.770585+00:00","updated_at":"2026-07-05T12:01:15.770585+00:00"}