{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RUOVSYBEFDJ3EUVEGKZ6C7RWTG","short_pith_number":"pith:RUOVSYBE","schema_version":"1.0","canonical_sha256":"8d1d59602428d3b252a432b3e17e3699a76bcbf34cbe4afac674d743530a64cb","source":{"kind":"arxiv","id":"2502.06428","version":2},"attestation_state":"computed","paper":{"title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyang Si, Jian Hu, Shaogang Gong, Wei Li, Zixu Cheng","submitted_at":"2025-02-10T13:03:05Z","abstract_excerpt":"Multi-modal Large Language Models (MLLMs) struggle with long videos due to the need for excessive visual tokens. These tokens exceed massively the context length of MLLMs, resulting in filled by redundant task-irrelevant shots. How to select shots is an unsolved critical problem: sparse sampling risks missing key details, while exhaustive sampling overwhelms the model with irrelevant content, leading to video misunderstanding. To solve this problem, we propose Chain-of-Shot prompting (CoS). The key idea is to frame shot selection as test-time visual prompt optimisation, choosing shots adaptive"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06428","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-10T13:03:05Z","cross_cats_sorted":[],"title_canon_sha256":"217cb83a9587c80dfc9725485a34722bf4d6b4bcc891d1032e388e584e6f0a6e","abstract_canon_sha256":"57cb6141dbe405ab9ca21b0c5b6a3c5af046dfe4ca62b88d4176f04c066a3d01"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:27.810856Z","signature_b64":"v1fbw1wveyN2UInfYRWj2VGUiGr67LERS05r2Kt2I4FHb/VxPw87ee0zYjuGiDjyxWJ4AOm7jYizwSgeV6KrCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d1d59602428d3b252a432b3e17e3699a76bcbf34cbe4afac674d743530a64cb","last_reissued_at":"2026-07-05T10:12:27.810414Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:27.810414Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyang Si, Jian Hu, Shaogang Gong, Wei Li, Zixu Cheng","submitted_at":"2025-02-10T13:03:05Z","abstract_excerpt":"Multi-modal Large Language Models (MLLMs) struggle with long videos due to the need for excessive visual tokens. These tokens exceed massively the context length of MLLMs, resulting in filled by redundant task-irrelevant shots. How to select shots is an unsolved critical problem: sparse sampling risks missing key details, while exhaustive sampling overwhelms the model with irrelevant content, leading to video misunderstanding. To solve this problem, we propose Chain-of-Shot prompting (CoS). The key idea is to frame shot selection as test-time visual prompt optimisation, choosing shots adaptive"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06428","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06428/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06428","created_at":"2026-07-05T10:12:27.810471+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06428v2","created_at":"2026-07-05T10:12:27.810471+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06428","created_at":"2026-07-05T10:12:27.810471+00:00"},{"alias_kind":"pith_short_12","alias_value":"RUOVSYBEFDJ3","created_at":"2026-07-05T10:12:27.810471+00:00"},{"alias_kind":"pith_short_16","alias_value":"RUOVSYBEFDJ3EUVE","created_at":"2026-07-05T10:12:27.810471+00:00"},{"alias_kind":"pith_short_8","alias_value":"RUOVSYBE","created_at":"2026-07-05T10:12:27.810471+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29445","citing_title":"Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22013","citing_title":"PointLLM-R: Enhancing 3D Point Cloud Reasoning via Chain-of-Thought","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22678","citing_title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01657","citing_title":"Act2See: Emergent Active Visual Perception for Video Reasoning","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG","json":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG.json","graph_json":"https://pith.science/api/pith-number/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/graph.json","events_json":"https://pith.science/api/pith-number/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/events.json","paper":"https://pith.science/paper/RUOVSYBE"},"agent_actions":{"view_html":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG","download_json":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG.json","view_paper":"https://pith.science/paper/RUOVSYBE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06428&json=true","fetch_graph":"https://pith.science/api/pith-number/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/graph.json","fetch_events":"https://pith.science/api/pith-number/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/action/storage_attestation","attest_author":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/action/author_attestation","sign_citation":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/action/citation_signature","submit_replication":"https://pith.science/pith/RUOVSYBEFDJ3EUVEGKZ6C7RWTG/action/replication_record"}},"created_at":"2026-07-05T10:12:27.810471+00:00","updated_at":"2026-07-05T10:12:27.810471+00:00"}