{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KYEWW72MLUODTEDFFIYMM6BFBR","short_pith_number":"pith:KYEWW72M","schema_version":"1.0","canonical_sha256":"56096b7f4c5d1c3990652a30c678250c5b79e21cfa903ecd40f171a19e0cb7ef","source":{"kind":"arxiv","id":"2503.20362","version":2},"attestation_state":"computed","paper":{"title":"Self-ReS: Self-Reflection in Large Vision-Language Models for Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"David Semedo, Joao Neves, Joao Pereira, Vasco Lopes","submitted_at":"2025-03-26T09:39:58Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) demonstrate remarkable performance in short-video tasks such as video question answering, but struggle in long-video understanding. The linear frame sampling strategy, conventionally used by LVLMs, fails to account for the non-linear distribution of key events in video data, often introducing redundant or irrelevant information in longer contexts while risking the omission of critical events in shorter ones. To address this, we propose SelfReS, a non-linear spatiotemporal self-reflective sampling method that dynamically selects key video fragments based on "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20362","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-26T09:39:58Z","cross_cats_sorted":[],"title_canon_sha256":"6423a6e8b7e89c4d74264f145e5a4370105d82e438aea30802a4e43fd6a9efe6","abstract_canon_sha256":"ffdc9b82e5707c0cade13b4d22fa94cd3dc67cb2c3006137eee2cabeb5d5b988"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:15.339288Z","signature_b64":"FQWeIdhydDjxlrxxebUEvzqhPotI24CBzj9dzcEHCPtatpM5ZOuQrZwpWTTfVu44fJBPn2LavIP6zNq3SyXEBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56096b7f4c5d1c3990652a30c678250c5b79e21cfa903ecd40f171a19e0cb7ef","last_reissued_at":"2026-07-05T11:28:15.338735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:15.338735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-ReS: Self-Reflection in Large Vision-Language Models for Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"David Semedo, Joao Neves, Joao Pereira, Vasco Lopes","submitted_at":"2025-03-26T09:39:58Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) demonstrate remarkable performance in short-video tasks such as video question answering, but struggle in long-video understanding. The linear frame sampling strategy, conventionally used by LVLMs, fails to account for the non-linear distribution of key events in video data, often introducing redundant or irrelevant information in longer contexts while risking the omission of critical events in shorter ones. To address this, we propose SelfReS, a non-linear spatiotemporal self-reflective sampling method that dynamically selects key video fragments based on "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20362","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20362/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20362","created_at":"2026-07-05T11:28:15.338789+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20362v2","created_at":"2026-07-05T11:28:15.338789+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20362","created_at":"2026-07-05T11:28:15.338789+00:00"},{"alias_kind":"pith_short_12","alias_value":"KYEWW72MLUOD","created_at":"2026-07-05T11:28:15.338789+00:00"},{"alias_kind":"pith_short_16","alias_value":"KYEWW72MLUODTEDF","created_at":"2026-07-05T11:28:15.338789+00:00"},{"alias_kind":"pith_short_8","alias_value":"KYEWW72M","created_at":"2026-07-05T11:28:15.338789+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR","json":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR.json","graph_json":"https://pith.science/api/pith-number/KYEWW72MLUODTEDFFIYMM6BFBR/graph.json","events_json":"https://pith.science/api/pith-number/KYEWW72MLUODTEDFFIYMM6BFBR/events.json","paper":"https://pith.science/paper/KYEWW72M"},"agent_actions":{"view_html":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR","download_json":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR.json","view_paper":"https://pith.science/paper/KYEWW72M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20362&json=true","fetch_graph":"https://pith.science/api/pith-number/KYEWW72MLUODTEDFFIYMM6BFBR/graph.json","fetch_events":"https://pith.science/api/pith-number/KYEWW72MLUODTEDFFIYMM6BFBR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR/action/storage_attestation","attest_author":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR/action/author_attestation","sign_citation":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR/action/citation_signature","submit_replication":"https://pith.science/pith/KYEWW72MLUODTEDFFIYMM6BFBR/action/replication_record"}},"created_at":"2026-07-05T11:28:15.338789+00:00","updated_at":"2026-07-05T11:28:15.338789+00:00"}