{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FKDMJBWWWXRFGZ6USCV3Q3BWDZ","short_pith_number":"pith:FKDMJBWW","schema_version":"1.0","canonical_sha256":"2a86c486d6b5e25367d490abb86c361e638379340fd41aaca4abfabdfb9458e6","source":{"kind":"arxiv","id":"2507.02001","version":1},"attestation_state":"computed","paper":{"title":"Temporal Chain of Thought: Long-Video Understanding by Thinking in Frames","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ahmet Iscen, Alireza Fathi, Anurag Arnab, Cordelia Schmid, Mathilde Caron","submitted_at":"2025-07-01T18:39:26Z","abstract_excerpt":"Despite recent advances in Vision-Language Models (VLMs), long-video understanding remains a challenging problem. Although state-of-the-art long-context VLMs can process around 1000 input frames, they still struggle to effectively leverage this sequence length, and succumb to irrelevant distractors within the context window. We present Temporal Chain of Thought, an inference strategy for video question-answering that curates the model's input context. We use the VLM itself to iteratively identify and extract the most relevant frames from the video, which are then used for answering. We demonst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.02001","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-01T18:39:26Z","cross_cats_sorted":[],"title_canon_sha256":"effec4488ee1f19668045184bce7ad4a9f246669de4224ad4cf46838e115941d","abstract_canon_sha256":"1ffb42227ea28207201765e7b52abd447defd31e9bee6a7eb403dde347d2427a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:10.281795Z","signature_b64":"YspVEU0G5XWSfkZGl4ev/d8LogRcbUPxFxsR4PQ08gkekRLna5GrgytDr/+VxCwXqUGBMBXvM0I38rnjg4mEBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a86c486d6b5e25367d490abb86c361e638379340fd41aaca4abfabdfb9458e6","last_reissued_at":"2026-07-05T11:31:10.281321Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:10.281321Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Temporal Chain of Thought: Long-Video Understanding by Thinking in Frames","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ahmet Iscen, Alireza Fathi, Anurag Arnab, Cordelia Schmid, Mathilde Caron","submitted_at":"2025-07-01T18:39:26Z","abstract_excerpt":"Despite recent advances in Vision-Language Models (VLMs), long-video understanding remains a challenging problem. Although state-of-the-art long-context VLMs can process around 1000 input frames, they still struggle to effectively leverage this sequence length, and succumb to irrelevant distractors within the context window. We present Temporal Chain of Thought, an inference strategy for video question-answering that curates the model's input context. We use the VLM itself to iteratively identify and extract the most relevant frames from the video, which are then used for answering. We demonst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.02001","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.02001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.02001","created_at":"2026-07-05T11:31:10.281387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.02001v1","created_at":"2026-07-05T11:31:10.281387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.02001","created_at":"2026-07-05T11:31:10.281387+00:00"},{"alias_kind":"pith_short_12","alias_value":"FKDMJBWWWXRF","created_at":"2026-07-05T11:31:10.281387+00:00"},{"alias_kind":"pith_short_16","alias_value":"FKDMJBWWWXRFGZ6U","created_at":"2026-07-05T11:31:10.281387+00:00"},{"alias_kind":"pith_short_8","alias_value":"FKDMJBWW","created_at":"2026-07-05T11:31:10.281387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03100","citing_title":"Zero-Shot 3D Question Answering via Hierarchical View-to-Token Transportation","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31734","citing_title":"MemLearner: Learning to Query Context memory for Video World Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22678","citing_title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02371","citing_title":"Internalized Reasoning for Long-Context Visual Document Understanding","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10936","citing_title":"Personal Visual Context Learning in Large Multimodal Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06185","citing_title":"Event-Causal RAG: A Retrieval-Augmented Generation Framework for Long Video Reasoning in Complex Scenarios","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ","json":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ.json","graph_json":"https://pith.science/api/pith-number/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/graph.json","events_json":"https://pith.science/api/pith-number/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/events.json","paper":"https://pith.science/paper/FKDMJBWW"},"agent_actions":{"view_html":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ","download_json":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ.json","view_paper":"https://pith.science/paper/FKDMJBWW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.02001&json=true","fetch_graph":"https://pith.science/api/pith-number/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/graph.json","fetch_events":"https://pith.science/api/pith-number/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/action/storage_attestation","attest_author":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/action/author_attestation","sign_citation":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/action/citation_signature","submit_replication":"https://pith.science/pith/FKDMJBWWWXRFGZ6USCV3Q3BWDZ/action/replication_record"}},"created_at":"2026-07-05T11:31:10.281387+00:00","updated_at":"2026-07-05T11:31:10.281387+00:00"}