{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GKMJZZPXCLFTIUOYVVWJNK3DDB","short_pith_number":"pith:GKMJZZPX","schema_version":"1.0","canonical_sha256":"32989ce5f712cb3451d8ad6c96ab6318514bfab2f932cc0dee93a9a9b8869af7","source":{"kind":"arxiv","id":"2304.14407","version":2},"attestation_state":"computed","paper":{"title":"ChatVideo: A Tracklet-centric Multimodal and Versatile Video Understanding System","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Luo, Dongdong Chen, Junke Wang, Lu Yuan, Xiyang Dai, Yu-Gang Jiang, Zuxuan Wu","submitted_at":"2023-04-27T17:59:58Z","abstract_excerpt":"Existing deep video models are limited by specific tasks, fixed input-output spaces, and poor generalization capabilities, making it difficult to deploy them in real-world scenarios. In this paper, we present our vision for multimodal and versatile video understanding and propose a prototype system, \\system. Our system is built upon a tracklet-centric paradigm, which treats tracklets as the basic video unit and employs various Video Foundation Models (ViFMs) to annotate their properties e.g., appearance, motion, \\etc. All the detected tracklets are stored in a database and interact with the us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.14407","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-04-27T17:59:58Z","cross_cats_sorted":[],"title_canon_sha256":"e5a84f2fd7e20dc54941fd1be6a008c960b4e295820206ef983d9a811d3e5889","abstract_canon_sha256":"937636bf35f9109e288c5ec2f5fb56cdada230e77d139630e949d9b1f4c3c269"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:05:36.196278Z","signature_b64":"oHOebvX7zIwhwHRT8pjPXgXIsIXEvGxdhHRaTnlpaVBQOxKxltzNfGq+MT6rKFd8uG3I+3/yURlbL5R261yFBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32989ce5f712cb3451d8ad6c96ab6318514bfab2f932cc0dee93a9a9b8869af7","last_reissued_at":"2026-07-05T06:05:36.195850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:05:36.195850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChatVideo: A Tracklet-centric Multimodal and Versatile Video Understanding System","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Luo, Dongdong Chen, Junke Wang, Lu Yuan, Xiyang Dai, Yu-Gang Jiang, Zuxuan Wu","submitted_at":"2023-04-27T17:59:58Z","abstract_excerpt":"Existing deep video models are limited by specific tasks, fixed input-output spaces, and poor generalization capabilities, making it difficult to deploy them in real-world scenarios. In this paper, we present our vision for multimodal and versatile video understanding and propose a prototype system, \\system. Our system is built upon a tracklet-centric paradigm, which treats tracklets as the basic video unit and employs various Video Foundation Models (ViFMs) to annotate their properties e.g., appearance, motion, \\etc. All the detected tracklets are stored in a database and interact with the us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.14407","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.14407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.14407","created_at":"2026-07-05T06:05:36.195905+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.14407v2","created_at":"2026-07-05T06:05:36.195905+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.14407","created_at":"2026-07-05T06:05:36.195905+00:00"},{"alias_kind":"pith_short_12","alias_value":"GKMJZZPXCLFT","created_at":"2026-07-05T06:05:36.195905+00:00"},{"alias_kind":"pith_short_16","alias_value":"GKMJZZPXCLFTIUOY","created_at":"2026-07-05T06:05:36.195905+00:00"},{"alias_kind":"pith_short_8","alias_value":"GKMJZZPX","created_at":"2026-07-05T06:05:36.195905+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2407.08101","citing_title":"What to Say and When to Say it: Live Fitness Coaching as a Testbed for Situated Interaction","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01400","citing_title":"Token Reduction via Local and Global Contexts Optimization for Efficient Video Large Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01455","citing_title":"From Verbatim to Gist: Distilling Pyramidal Multimodal Memory via Semantic Information Bottleneck for Long-Horizon Video Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07897","citing_title":"Semantic-Aware Adaptive Visual Memory for Streaming Video Understanding","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB","json":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB.json","graph_json":"https://pith.science/api/pith-number/GKMJZZPXCLFTIUOYVVWJNK3DDB/graph.json","events_json":"https://pith.science/api/pith-number/GKMJZZPXCLFTIUOYVVWJNK3DDB/events.json","paper":"https://pith.science/paper/GKMJZZPX"},"agent_actions":{"view_html":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB","download_json":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB.json","view_paper":"https://pith.science/paper/GKMJZZPX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.14407&json=true","fetch_graph":"https://pith.science/api/pith-number/GKMJZZPXCLFTIUOYVVWJNK3DDB/graph.json","fetch_events":"https://pith.science/api/pith-number/GKMJZZPXCLFTIUOYVVWJNK3DDB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB/action/storage_attestation","attest_author":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB/action/author_attestation","sign_citation":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB/action/citation_signature","submit_replication":"https://pith.science/pith/GKMJZZPXCLFTIUOYVVWJNK3DDB/action/replication_record"}},"created_at":"2026-07-05T06:05:36.195905+00:00","updated_at":"2026-07-05T06:05:36.195905+00:00"}