{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SNGQJFUGLIY5ITA26PDZBOF4OL","short_pith_number":"pith:SNGQJFUG","schema_version":"1.0","canonical_sha256":"934d0496865a31d44c1af3c790b8bc72dd633288d38c3d6747faafb4b0c8ab3c","source":{"kind":"arxiv","id":"2507.09876","version":1},"attestation_state":"computed","paper":{"title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Hao Fei, Libo Qin, Qiguang Chen, Ruihan Tao, Wanxiang Che, Xu Liu, Yongheng Zhang","submitted_at":"2025-07-14T03:21:13Z","abstract_excerpt":"Video understanding plays a vital role in bridging low-level visual signals with high-level cognitive reasoning, and is fundamental to applications such as autonomous driving, embodied AI, and the broader pursuit of AGI. The rapid development of large language models (LLMs), particularly those utilizing Chain-of-Thought (CoT) technology, has significantly advanced video reasoning capabilities. However, current approaches primarily depend on textual information for reasoning, overlooking the visual modality in the actual video reasoning process. In contrast, humans naturally re-examine visual c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09876","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-14T03:21:13Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5fe378c1b8a6e061564a2e183159a577a7909962f2653763e7a706b62c512841","abstract_canon_sha256":"593bbb83a3bf47d6c8e6fd870c8f9861462b1eaab811877248499a2752a6c798"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:41.457751Z","signature_b64":"12R3k1VbbIgiZvx5dzdTTp+7fC4nIKx5DmLY244e4/PTf8W9fRwdMOfF/bRyvz7Hfm0ogiq+/uyKsz6s/aS+Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"934d0496865a31d44c1af3c790b8bc72dd633288d38c3d6747faafb4b0c8ab3c","last_reissued_at":"2026-07-05T11:36:41.457246Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:41.457246Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Hao Fei, Libo Qin, Qiguang Chen, Ruihan Tao, Wanxiang Che, Xu Liu, Yongheng Zhang","submitted_at":"2025-07-14T03:21:13Z","abstract_excerpt":"Video understanding plays a vital role in bridging low-level visual signals with high-level cognitive reasoning, and is fundamental to applications such as autonomous driving, embodied AI, and the broader pursuit of AGI. The rapid development of large language models (LLMs), particularly those utilizing Chain-of-Thought (CoT) technology, has significantly advanced video reasoning capabilities. However, current approaches primarily depend on textual information for reasoning, overlooking the visual modality in the actual video reasoning process. In contrast, humans naturally re-examine visual c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09876","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09876/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09876","created_at":"2026-07-05T11:36:41.457307+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09876v1","created_at":"2026-07-05T11:36:41.457307+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09876","created_at":"2026-07-05T11:36:41.457307+00:00"},{"alias_kind":"pith_short_12","alias_value":"SNGQJFUGLIY5","created_at":"2026-07-05T11:36:41.457307+00:00"},{"alias_kind":"pith_short_16","alias_value":"SNGQJFUGLIY5ITA2","created_at":"2026-07-05T11:36:41.457307+00:00"},{"alias_kind":"pith_short_8","alias_value":"SNGQJFUG","created_at":"2026-07-05T11:36:41.457307+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05736","citing_title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17283","citing_title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04570","citing_title":"Thinking with Video: Video Generation as a Promising Multimodal Reasoning Paradigm","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01657","citing_title":"Act2See: Emergent Active Visual Perception for Video Reasoning","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL","json":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL.json","graph_json":"https://pith.science/api/pith-number/SNGQJFUGLIY5ITA26PDZBOF4OL/graph.json","events_json":"https://pith.science/api/pith-number/SNGQJFUGLIY5ITA26PDZBOF4OL/events.json","paper":"https://pith.science/paper/SNGQJFUG"},"agent_actions":{"view_html":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL","download_json":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL.json","view_paper":"https://pith.science/paper/SNGQJFUG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09876&json=true","fetch_graph":"https://pith.science/api/pith-number/SNGQJFUGLIY5ITA26PDZBOF4OL/graph.json","fetch_events":"https://pith.science/api/pith-number/SNGQJFUGLIY5ITA26PDZBOF4OL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL/action/storage_attestation","attest_author":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL/action/author_attestation","sign_citation":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL/action/citation_signature","submit_replication":"https://pith.science/pith/SNGQJFUGLIY5ITA26PDZBOF4OL/action/replication_record"}},"created_at":"2026-07-05T11:36:41.457307+00:00","updated_at":"2026-07-05T11:36:41.457307+00:00"}