{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KSYAV4EBYKYO5LNKVHLB4C2UM7","short_pith_number":"pith:KSYAV4EB","schema_version":"1.0","canonical_sha256":"54b00af081c2b0eeadaaa9d61e0b5467f29535cc1aedf4440422816a015f8dca","source":{"kind":"arxiv","id":"2407.05355","version":1},"attestation_state":"computed","paper":{"title":"VideoCoT: A Video Chain-of-Thought Dataset with Active Annotation Tool","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jingsheng Zheng, Jin Xu, Xiangmin Xu, Xiaofen Xing, Yan Wang, Yawen Zeng","submitted_at":"2024-07-07T13:10:23Z","abstract_excerpt":"Multimodal large language models (MLLMs) are flourishing, but mainly focus on images with less attention than videos, especially in sub-fields such as prompt engineering, video chain-of-thought (CoT), and instruction tuning on videos. Therefore, we try to explore the collection of CoT datasets in videos to lead to video OpenQA and improve the reasoning ability of MLLMs. Unfortunately, making such video CoT datasets is not an easy task. Given that human annotation is too cumbersome and expensive, while machine-generated is not reliable due to the hallucination issue, we develop an automatic ann"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05355","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-07T13:10:23Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f5ffaadf10636b5d0d2bfb7b3eb3bd8234c0ee8e8ce53fe4d7b35c2a78288995","abstract_canon_sha256":"865e25aa7072fa0494ebe2a81c683ab0ad761e75d17c0e959fe4206d9f4db03c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:11.946498Z","signature_b64":"KiOWUspNxNgSaqJ8BvHCkdBbOoB4FqWpi0IFeeLFChp+w/JhFxYji3qQwdISsqUHZpB1XqwAd5kVk7OysYQQDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54b00af081c2b0eeadaaa9d61e0b5467f29535cc1aedf4440422816a015f8dca","last_reissued_at":"2026-07-05T08:41:11.946048Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:11.946048Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoCoT: A Video Chain-of-Thought Dataset with Active Annotation Tool","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jingsheng Zheng, Jin Xu, Xiangmin Xu, Xiaofen Xing, Yan Wang, Yawen Zeng","submitted_at":"2024-07-07T13:10:23Z","abstract_excerpt":"Multimodal large language models (MLLMs) are flourishing, but mainly focus on images with less attention than videos, especially in sub-fields such as prompt engineering, video chain-of-thought (CoT), and instruction tuning on videos. Therefore, we try to explore the collection of CoT datasets in videos to lead to video OpenQA and improve the reasoning ability of MLLMs. Unfortunately, making such video CoT datasets is not an easy task. Given that human annotation is too cumbersome and expensive, while machine-generated is not reliable due to the hallucination issue, we develop an automatic ann"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05355","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05355/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05355","created_at":"2026-07-05T08:41:11.946100+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05355v1","created_at":"2026-07-05T08:41:11.946100+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05355","created_at":"2026-07-05T08:41:11.946100+00:00"},{"alias_kind":"pith_short_12","alias_value":"KSYAV4EBYKYO","created_at":"2026-07-05T08:41:11.946100+00:00"},{"alias_kind":"pith_short_16","alias_value":"KSYAV4EBYKYO5LNK","created_at":"2026-07-05T08:41:11.946100+00:00"},{"alias_kind":"pith_short_8","alias_value":"KSYAV4EB","created_at":"2026-07-05T08:41:11.946100+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23216","citing_title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":285,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05736","citing_title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23216","citing_title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15342","citing_title":"Minerva-Ego: Spatiotemporal Hints for Egocentric Video Understanding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":159,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01657","citing_title":"Act2See: Emergent Active Visual Perception for Video Reasoning","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7","json":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7.json","graph_json":"https://pith.science/api/pith-number/KSYAV4EBYKYO5LNKVHLB4C2UM7/graph.json","events_json":"https://pith.science/api/pith-number/KSYAV4EBYKYO5LNKVHLB4C2UM7/events.json","paper":"https://pith.science/paper/KSYAV4EB"},"agent_actions":{"view_html":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7","download_json":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7.json","view_paper":"https://pith.science/paper/KSYAV4EB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05355&json=true","fetch_graph":"https://pith.science/api/pith-number/KSYAV4EBYKYO5LNKVHLB4C2UM7/graph.json","fetch_events":"https://pith.science/api/pith-number/KSYAV4EBYKYO5LNKVHLB4C2UM7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7/action/storage_attestation","attest_author":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7/action/author_attestation","sign_citation":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7/action/citation_signature","submit_replication":"https://pith.science/pith/KSYAV4EBYKYO5LNKVHLB4C2UM7/action/replication_record"}},"created_at":"2026-07-05T08:41:11.946100+00:00","updated_at":"2026-07-05T08:41:11.946100+00:00"}