{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QMFIVEHEOTUNQYUKBPO7SOAE4W","short_pith_number":"pith:QMFIVEHE","schema_version":"1.0","canonical_sha256":"830a8a90e474e8d8628a0bddf93804e58808ea195f72f6ad866347161b86e228","source":{"kind":"arxiv","id":"2306.07207","version":3},"attestation_state":"computed","paper":{"title":"Valley: Video Assistant with Large Language model Enhanced abilitY","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Cen Chen, Minghui Qiu, Min Yang, Ruipu Luo, Tao Wang, Yanhao Wang, Zheming Yang, Zhongyu Wei, Ziwang Zhao","submitted_at":"2023-06-12T16:11:10Z","abstract_excerpt":"Large Language Models (LLMs), with remarkable conversational capability, have emerged as AI assistants that can handle both visual and textual modalities. However, their effectiveness in joint video and language understanding has not been extensively explored. In the paper, we introduce Valley, a multi-modal foundation model that is designed to enable enhanced video comprehension and instruction-following capabilities. To this end, we construct two datasets, namely Valley-702k and Valley-instruct-73k, to cover a diverse range of video-text alignment and video-based instruction tasks, such as m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.07207","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-12T16:11:10Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"1105366a330eb94b3ec402c228006df5045596d65a1d5dbafa7a5f4d77c909f8","abstract_canon_sha256":"4fc310cd6bbf865a970b88cd263b25aad714f59e9bed354a0b6ba77a5d032dcd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:23.291528Z","signature_b64":"m4APRj7WU6iPBNSuc4Y59wJMntTnJFRgeIdK5bTjuiaWU+zr/TBbQoYpiZCIGobkFzRjq6Kkl52WHbeUIdlpCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"830a8a90e474e8d8628a0bddf93804e58808ea195f72f6ad866347161b86e228","last_reissued_at":"2026-07-05T10:32:23.291043Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:23.291043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Valley: Video Assistant with Large Language model Enhanced abilitY","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Cen Chen, Minghui Qiu, Min Yang, Ruipu Luo, Tao Wang, Yanhao Wang, Zheming Yang, Zhongyu Wei, Ziwang Zhao","submitted_at":"2023-06-12T16:11:10Z","abstract_excerpt":"Large Language Models (LLMs), with remarkable conversational capability, have emerged as AI assistants that can handle both visual and textual modalities. However, their effectiveness in joint video and language understanding has not been extensively explored. In the paper, we introduce Valley, a multi-modal foundation model that is designed to enable enhanced video comprehension and instruction-following capabilities. To this end, we construct two datasets, namely Valley-702k and Valley-instruct-73k, to cover a diverse range of video-text alignment and video-based instruction tasks, such as m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.07207","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.07207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.07207","created_at":"2026-07-05T10:32:23.291106+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.07207v3","created_at":"2026-07-05T10:32:23.291106+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.07207","created_at":"2026-07-05T10:32:23.291106+00:00"},{"alias_kind":"pith_short_12","alias_value":"QMFIVEHEOTUN","created_at":"2026-07-05T10:32:23.291106+00:00"},{"alias_kind":"pith_short_16","alias_value":"QMFIVEHEOTUNQYUK","created_at":"2026-07-05T10:32:23.291106+00:00"},{"alias_kind":"pith_short_8","alias_value":"QMFIVEHE","created_at":"2026-07-05T10:32:23.291106+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25325","citing_title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22352","citing_title":"On the Sparsity-Storage-Accuracy Tradeoff in Parsimoniously Activated Dictionary Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10533","citing_title":"Audio-Visual Exchange-Aware Token Pruning for Efficient Audio-Visual Captioning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07289","citing_title":"Closed-Form Spectral Regularization for Multi-Task Model Merging","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01117","citing_title":"MoHallBench: A Benchmark for Motion Hallucination in Video Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2411.02327","citing_title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2412.02930","citing_title":"TemporalVLM: Video LLMs for Temporal Reasoning in Long Videos","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09158","citing_title":"FaVChat: Hierarchical Prompt-Query Guided Facial Video Understanding with Data-Efficient GRPO","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18904","citing_title":"Dynamic Model Merging Made Slim","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18962","citing_title":"UniMind: Unleashing the Power of LLMs for Unified Multi-Task Brain Decoding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2311.17005","citing_title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2403.00476","citing_title":"TempCompass: Do Video LLMs Really Understand Videos?","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06673","citing_title":"Detector-Empowered Video Large Language Model for Efficient Spatio-Temporal Grounding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17434","citing_title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2311.10122","citing_title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2307.16125","citing_title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27968","citing_title":"ClimateVID -- Social Media Videos Analysis and Challenges Involved","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05079","citing_title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W","json":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W.json","graph_json":"https://pith.science/api/pith-number/QMFIVEHEOTUNQYUKBPO7SOAE4W/graph.json","events_json":"https://pith.science/api/pith-number/QMFIVEHEOTUNQYUKBPO7SOAE4W/events.json","paper":"https://pith.science/paper/QMFIVEHE"},"agent_actions":{"view_html":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W","download_json":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W.json","view_paper":"https://pith.science/paper/QMFIVEHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.07207&json=true","fetch_graph":"https://pith.science/api/pith-number/QMFIVEHEOTUNQYUKBPO7SOAE4W/graph.json","fetch_events":"https://pith.science/api/pith-number/QMFIVEHEOTUNQYUKBPO7SOAE4W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W/action/storage_attestation","attest_author":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W/action/author_attestation","sign_citation":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W/action/citation_signature","submit_replication":"https://pith.science/pith/QMFIVEHEOTUNQYUKBPO7SOAE4W/action/replication_record"}},"created_at":"2026-07-05T10:32:23.291106+00:00","updated_at":"2026-07-05T10:32:23.291106+00:00"}