{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:NU2OVGHAQRDPMSGEIW3TR4P4KA","short_pith_number":"pith:NU2OVGHA","schema_version":"1.0","canonical_sha256":"6d34ea98e08446f648c445b738f1fc500e8a69b07f63e4443b027fa9b2e96c46","source":{"kind":"arxiv","id":"2608.05592","version":1},"attestation_state":"computed","paper":{"title":"Beyond Frame Selection: Rethinking Long-Video Understanding with MLLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Shin'ichi Satoh, Ziling Huang","submitted_at":"2026-08-06T04:27:59Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved strong progress in video understanding, yet it remains challenging because the token limitation makes MLLMs difficult to capture temporally sparse evidence. Existing methods typically rely on uniform sampling, or frame selection, but these strategies usually optimize either broad temporal coverage or local relevance, making it difficult to preserve both global storyline context and fine-grained evidence. We propose VideoRouter(VR) that rethinks long-video understanding as coordinating complementary evidence views rather than selecting a si"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.05592","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-08-06T04:27:59Z","cross_cats_sorted":[],"title_canon_sha256":"b585d573daa10390b9b99d209724d3df1b9da75f8907b4fa19306a2b8914b5d3","abstract_canon_sha256":"2a0eaf0b4fc1bbcd18ae1986b73dc821c9819679d2be8e2464bbf92038fc0b66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-07T00:50:15.468015Z","signature_b64":"dNs5CWDPkgSlxWA4XBV3Wig8TIZ+WJhCd2zoyauoAn2wRamSo5SRBf+SySs5Xywxm9GazpkPK0aDvv2G2REoCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d34ea98e08446f648c445b738f1fc500e8a69b07f63e4443b027fa9b2e96c46","last_reissued_at":"2026-08-07T00:50:15.466603Z","signature_status":"signed_v1","first_computed_at":"2026-08-07T00:50:15.466603Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Frame Selection: Rethinking Long-Video Understanding with MLLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Shin'ichi Satoh, Ziling Huang","submitted_at":"2026-08-06T04:27:59Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved strong progress in video understanding, yet it remains challenging because the token limitation makes MLLMs difficult to capture temporally sparse evidence. Existing methods typically rely on uniform sampling, or frame selection, but these strategies usually optimize either broad temporal coverage or local relevance, making it difficult to preserve both global storyline context and fine-grained evidence. We propose VideoRouter(VR) that rethinks long-video understanding as coordinating complementary evidence views rather than selecting a si"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.05592","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.05592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.05592","created_at":"2026-08-07T00:50:15.468036+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.05592v1","created_at":"2026-08-07T00:50:15.468036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.05592","created_at":"2026-08-07T00:50:15.468036+00:00"},{"alias_kind":"pith_short_12","alias_value":"NU2OVGHAQRDP","created_at":"2026-08-07T00:50:15.468036+00:00"},{"alias_kind":"pith_short_16","alias_value":"NU2OVGHAQRDPMSGE","created_at":"2026-08-07T00:50:15.468036+00:00"},{"alias_kind":"pith_short_8","alias_value":"NU2OVGHA","created_at":"2026-08-07T00:50:15.468036+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA","json":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA.json","graph_json":"https://pith.science/api/pith-number/NU2OVGHAQRDPMSGEIW3TR4P4KA/graph.json","events_json":"https://pith.science/api/pith-number/NU2OVGHAQRDPMSGEIW3TR4P4KA/events.json","paper":"https://pith.science/paper/NU2OVGHA"},"agent_actions":{"view_html":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA","download_json":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA.json","view_paper":"https://pith.science/paper/NU2OVGHA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.05592&json=true","fetch_graph":"https://pith.science/api/pith-number/NU2OVGHAQRDPMSGEIW3TR4P4KA/graph.json","fetch_events":"https://pith.science/api/pith-number/NU2OVGHAQRDPMSGEIW3TR4P4KA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA/action/storage_attestation","attest_author":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA/action/author_attestation","sign_citation":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA/action/citation_signature","submit_replication":"https://pith.science/pith/NU2OVGHAQRDPMSGEIW3TR4P4KA/action/replication_record"}},"created_at":"2026-08-07T00:50:15.468036+00:00","updated_at":"2026-08-07T00:50:15.468036+00:00"}