{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GMMFBDO5JSWFYMVUOSSVAJUNYN","short_pith_number":"pith:GMMFBDO5","schema_version":"1.0","canonical_sha256":"3318508ddd4cac5c32b474a550268dc35e0a0fab8739e55a58c1cc2b42892671","source":{"kind":"arxiv","id":"2507.00033","version":1},"attestation_state":"computed","paper":{"title":"Moment Sampling in Video LLMs for Long-Form Video QA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Andrea Fanelli, Gauri Jagatap, Gouthaman KV, Grant Van Horn, Mustafa Chasmai, Subhransu Maji","submitted_at":"2025-06-18T03:23:56Z","abstract_excerpt":"Recent advancements in video large language models (Video LLMs) have significantly advanced the field of video question answering (VideoQA). While existing methods perform well on short videos, they often struggle with long-range reasoning in longer videos. To scale Video LLMs for longer video content, frame sub-sampling (selecting frames at regular intervals) is commonly used. However, this approach is suboptimal, often leading to the loss of crucial frames or the inclusion of redundant information from multiple similar frames. Missing key frames impairs the model's ability to answer question"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.00033","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-18T03:23:56Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"01778b0bf95b5517142d42c0f32081bbd96a4fce00604597ed99abbe51cf76a9","abstract_canon_sha256":"b41e4212204c7f3e6fde9fc36c0fd176e802fbed50b0b5650ae9de2cb31de0f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:35.426717Z","signature_b64":"SOaIbBAMlUnEd275FfVYNiReZUp+Jw8TZjv7KKq+LqB/jWvFKy43SOSbH3lfKlc/ybCoPU5eFNOEfD/KL5zZAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3318508ddd4cac5c32b474a550268dc35e0a0fab8739e55a58c1cc2b42892671","last_reissued_at":"2026-07-05T11:29:35.426200Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:35.426200Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Moment Sampling in Video LLMs for Long-Form Video QA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Andrea Fanelli, Gauri Jagatap, Gouthaman KV, Grant Van Horn, Mustafa Chasmai, Subhransu Maji","submitted_at":"2025-06-18T03:23:56Z","abstract_excerpt":"Recent advancements in video large language models (Video LLMs) have significantly advanced the field of video question answering (VideoQA). While existing methods perform well on short videos, they often struggle with long-range reasoning in longer videos. To scale Video LLMs for longer video content, frame sub-sampling (selecting frames at regular intervals) is commonly used. However, this approach is suboptimal, often leading to the loss of crucial frames or the inclusion of redundant information from multiple similar frames. Missing key frames impairs the model's ability to answer question"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00033","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00033/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.00033","created_at":"2026-07-05T11:29:35.426265+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.00033v1","created_at":"2026-07-05T11:29:35.426265+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00033","created_at":"2026-07-05T11:29:35.426265+00:00"},{"alias_kind":"pith_short_12","alias_value":"GMMFBDO5JSWF","created_at":"2026-07-05T11:29:35.426265+00:00"},{"alias_kind":"pith_short_16","alias_value":"GMMFBDO5JSWFYMVU","created_at":"2026-07-05T11:29:35.426265+00:00"},{"alias_kind":"pith_short_8","alias_value":"GMMFBDO5","created_at":"2026-07-05T11:29:35.426265+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13141","citing_title":"Rethinking RAG in Long Videos: What to Retrieve and How to Use It?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04323","citing_title":"Answer Self-Consistency with Margin-Triggered Question Re-Arbitration for the CVPR 2026 VidLLMs Challenge","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05917","citing_title":"MemoryCard: Topic-Aware Multi-Modal Clue Compression for Long-Video Question Answering","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11113","citing_title":"VIDEOP2R: Video Understanding from Perception to Reasoning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN","json":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN.json","graph_json":"https://pith.science/api/pith-number/GMMFBDO5JSWFYMVUOSSVAJUNYN/graph.json","events_json":"https://pith.science/api/pith-number/GMMFBDO5JSWFYMVUOSSVAJUNYN/events.json","paper":"https://pith.science/paper/GMMFBDO5"},"agent_actions":{"view_html":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN","download_json":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN.json","view_paper":"https://pith.science/paper/GMMFBDO5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.00033&json=true","fetch_graph":"https://pith.science/api/pith-number/GMMFBDO5JSWFYMVUOSSVAJUNYN/graph.json","fetch_events":"https://pith.science/api/pith-number/GMMFBDO5JSWFYMVUOSSVAJUNYN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN/action/storage_attestation","attest_author":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN/action/author_attestation","sign_citation":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN/action/citation_signature","submit_replication":"https://pith.science/pith/GMMFBDO5JSWFYMVUOSSVAJUNYN/action/replication_record"}},"created_at":"2026-07-05T11:29:35.426265+00:00","updated_at":"2026-07-05T11:29:35.426265+00:00"}