{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F452E74KLC66BDEXBPBKXTL2DU","short_pith_number":"pith:F452E74K","schema_version":"1.0","canonical_sha256":"2f3ba27f8a58bde08c970bc2abcd7a1d3dced8f6135f679a8a35479346df46d3","source":{"kind":"arxiv","id":"2507.09491","version":1},"attestation_state":"computed","paper":{"title":"GLIMPSE: Do Large Vision-Language Models Truly Think With Videos or Just Glimpse at Them?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haibo Tong, Haonian Ji, Huaxiu Yao, Kangqi Li, Lijuan Wang, Linjie Li, Shi Qiu, Siwei Han, Yangfan He, Yiyang Zhou, Yuyang Zhao, Zhengyuan Yang, Zihao Zhao","submitted_at":"2025-07-13T04:44:57Z","abstract_excerpt":"Existing video benchmarks often resemble image-based benchmarks, with question types like \"What actions does the person perform throughout the video?\" or \"What color is the woman's dress in the video?\" For these, models can often answer by scanning just a few key frames, without deep temporal reasoning. This limits our ability to assess whether large vision-language models (LVLMs) can truly think with videos rather than perform superficial frame-level analysis. To address this, we introduce GLIMPSE, a benchmark specifically designed to evaluate whether LVLMs can genuinely think with videos. Un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09491","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-13T04:44:57Z","cross_cats_sorted":[],"title_canon_sha256":"8045a2be348f980ae33f176a907ac0bea90fdb753dfbc65377957b8a1ef7fe9e","abstract_canon_sha256":"b05b65871a2f811c005a12fdf9a7d460d3d552958303f20d3944f8e604ee8b4b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:09.983792Z","signature_b64":"xTS5919PwTO7NTlNU0mf3pnsn5OhagvdFqT0JjF4WFoC0O3NgDWZSv+OKizT5VBdbayFj1fp9F4UZO4/LuTsDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f3ba27f8a58bde08c970bc2abcd7a1d3dced8f6135f679a8a35479346df46d3","last_reissued_at":"2026-07-05T11:36:09.983305Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:09.983305Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GLIMPSE: Do Large Vision-Language Models Truly Think With Videos or Just Glimpse at Them?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haibo Tong, Haonian Ji, Huaxiu Yao, Kangqi Li, Lijuan Wang, Linjie Li, Shi Qiu, Siwei Han, Yangfan He, Yiyang Zhou, Yuyang Zhao, Zhengyuan Yang, Zihao Zhao","submitted_at":"2025-07-13T04:44:57Z","abstract_excerpt":"Existing video benchmarks often resemble image-based benchmarks, with question types like \"What actions does the person perform throughout the video?\" or \"What color is the woman's dress in the video?\" For these, models can often answer by scanning just a few key frames, without deep temporal reasoning. This limits our ability to assess whether large vision-language models (LVLMs) can truly think with videos rather than perform superficial frame-level analysis. To address this, we introduce GLIMPSE, a benchmark specifically designed to evaluate whether LVLMs can genuinely think with videos. Un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09491","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09491/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09491","created_at":"2026-07-05T11:36:09.983376+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09491v1","created_at":"2026-07-05T11:36:09.983376+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09491","created_at":"2026-07-05T11:36:09.983376+00:00"},{"alias_kind":"pith_short_12","alias_value":"F452E74KLC66","created_at":"2026-07-05T11:36:09.983376+00:00"},{"alias_kind":"pith_short_16","alias_value":"F452E74KLC66BDEX","created_at":"2026-07-05T11:36:09.983376+00:00"},{"alias_kind":"pith_short_8","alias_value":"F452E74K","created_at":"2026-07-05T11:36:09.983376+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU","json":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU.json","graph_json":"https://pith.science/api/pith-number/F452E74KLC66BDEXBPBKXTL2DU/graph.json","events_json":"https://pith.science/api/pith-number/F452E74KLC66BDEXBPBKXTL2DU/events.json","paper":"https://pith.science/paper/F452E74K"},"agent_actions":{"view_html":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU","download_json":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU.json","view_paper":"https://pith.science/paper/F452E74K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09491&json=true","fetch_graph":"https://pith.science/api/pith-number/F452E74KLC66BDEXBPBKXTL2DU/graph.json","fetch_events":"https://pith.science/api/pith-number/F452E74KLC66BDEXBPBKXTL2DU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU/action/storage_attestation","attest_author":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU/action/author_attestation","sign_citation":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU/action/citation_signature","submit_replication":"https://pith.science/pith/F452E74KLC66BDEXBPBKXTL2DU/action/replication_record"}},"created_at":"2026-07-05T11:36:09.983376+00:00","updated_at":"2026-07-05T11:36:09.983376+00:00"}