{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WFJ7BPBVX36ITKWFOYLH4ZOVWL","short_pith_number":"pith:WFJ7BPBV","schema_version":"1.0","canonical_sha256":"b153f0bc35befc89aac576167e65d5b2ebaaa96386171d999044e2868b805c9e","source":{"kind":"arxiv","id":"2304.04227","version":3},"attestation_state":"computed","paper":{"title":"Video ChatCaptioner: Towards Enriched Spatiotemporal Descriptions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Deyao Zhu, Jun Chen, Kilichbek Haydarov, Mohamed Elhoseiny, Xiang Li","submitted_at":"2023-04-09T12:46:18Z","abstract_excerpt":"Video captioning aims to convey dynamic scenes from videos using natural language, facilitating the understanding of spatiotemporal information within our environment. Although there have been recent advances, generating detailed and enriched video descriptions continues to be a substantial challenge. In this work, we introduce Video ChatCaptioner, an innovative approach for creating more comprehensive spatiotemporal video descriptions. Our method employs a ChatGPT model as a controller, specifically designed to select frames for posing video content-driven questions. Subsequently, a robust al"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.04227","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-04-09T12:46:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b5a37d60004c17964681e8e4a051459b97428d4c8d8b070ae16a5a3814314e8a","abstract_canon_sha256":"0864e89a420c27f97f4cd76ebbd0da0342d6b43f0e7e223119fcce4d84f79410"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:20.334111Z","signature_b64":"l6Xc/2MS1HeOs8y0u5Go+cA5RyO1+MKVOxu9PRHAKF4nLlI/ydgWeq52acbJTIqyZJJxzUjZ4RlTE5LWYfHiBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b153f0bc35befc89aac576167e65d5b2ebaaa96386171d999044e2868b805c9e","last_reissued_at":"2026-07-05T06:13:20.333686Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:20.333686Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video ChatCaptioner: Towards Enriched Spatiotemporal Descriptions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Deyao Zhu, Jun Chen, Kilichbek Haydarov, Mohamed Elhoseiny, Xiang Li","submitted_at":"2023-04-09T12:46:18Z","abstract_excerpt":"Video captioning aims to convey dynamic scenes from videos using natural language, facilitating the understanding of spatiotemporal information within our environment. Although there have been recent advances, generating detailed and enriched video descriptions continues to be a substantial challenge. In this work, we introduce Video ChatCaptioner, an innovative approach for creating more comprehensive spatiotemporal video descriptions. Our method employs a ChatGPT model as a controller, specifically designed to select frames for posing video content-driven questions. Subsequently, a robust al"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.04227","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.04227/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.04227","created_at":"2026-07-05T06:13:20.333747+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.04227v3","created_at":"2026-07-05T06:13:20.333747+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.04227","created_at":"2026-07-05T06:13:20.333747+00:00"},{"alias_kind":"pith_short_12","alias_value":"WFJ7BPBVX36I","created_at":"2026-07-05T06:13:20.333747+00:00"},{"alias_kind":"pith_short_16","alias_value":"WFJ7BPBVX36ITKWF","created_at":"2026-07-05T06:13:20.333747+00:00"},{"alias_kind":"pith_short_8","alias_value":"WFJ7BPBV","created_at":"2026-07-05T06:13:20.333747+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17423","citing_title":"Soap2Soap: Long Cinematic Video Remaking via Multi-Agent Collaboration","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2310.09478","citing_title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2304.10592","citing_title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL","json":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL.json","graph_json":"https://pith.science/api/pith-number/WFJ7BPBVX36ITKWFOYLH4ZOVWL/graph.json","events_json":"https://pith.science/api/pith-number/WFJ7BPBVX36ITKWFOYLH4ZOVWL/events.json","paper":"https://pith.science/paper/WFJ7BPBV"},"agent_actions":{"view_html":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL","download_json":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL.json","view_paper":"https://pith.science/paper/WFJ7BPBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.04227&json=true","fetch_graph":"https://pith.science/api/pith-number/WFJ7BPBVX36ITKWFOYLH4ZOVWL/graph.json","fetch_events":"https://pith.science/api/pith-number/WFJ7BPBVX36ITKWFOYLH4ZOVWL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL/action/storage_attestation","attest_author":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL/action/author_attestation","sign_citation":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL/action/citation_signature","submit_replication":"https://pith.science/pith/WFJ7BPBVX36ITKWFOYLH4ZOVWL/action/replication_record"}},"created_at":"2026-07-05T06:13:20.333747+00:00","updated_at":"2026-07-05T06:13:20.333747+00:00"}