{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AUMPF7GYLRMF4R5LJKAGHYFXIS","short_pith_number":"pith:AUMPF7GY","schema_version":"1.0","canonical_sha256":"0518f2fcd85c585e47ab4a8063e0b744ba1db8521964c46c4db63a4d01210bca","source":{"kind":"arxiv","id":"2407.14500","version":3},"attestation_state":"computed","paper":{"title":"ViLLa: Video Reasoning Segmentation with Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hengshuang Zhao, Kun Wang, Lu Qi, Rongkun Zheng, Xi Chen, Yi Wang, Yu Qiao","submitted_at":"2024-07-18T17:59:17Z","abstract_excerpt":"Recent efforts in video reasoning segmentation (VRS) integrate large language models (LLMs) with perception models to localize and track objects via textual instructions, achieving barely satisfactory results in simple scenarios. However, they struggled to discriminate and deduce the objects from user queries in more real-world scenes featured by long durations, multiple objects, rapid motion, and heavy occlusions. In this work, we analyze the underlying causes of these limitations, and present ViLLa: Video reasoning segmentation with Large Language Model. Remarkably, our ViLLa manages to tack"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.14500","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-18T17:59:17Z","cross_cats_sorted":[],"title_canon_sha256":"464e1c42a392829d04f2ddaa20640ac08749b4285fcf727f0f4d453942fbc241","abstract_canon_sha256":"0c68390b86a44a1dbd94945cffb89458013e1c1420b70fd31f0b4506fe6f1891"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:45.276056Z","signature_b64":"Td5sE7+izs0ArwxsSjYA2+8X7im2sDNHP2arxsuYW8V1z/J6jeSYKBM9uy/UM0Bg1ayFOj93IYPkPYVyuyfkCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0518f2fcd85c585e47ab4a8063e0b744ba1db8521964c46c4db63a4d01210bca","last_reissued_at":"2026-07-05T10:31:45.275536Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:45.275536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViLLa: Video Reasoning Segmentation with Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hengshuang Zhao, Kun Wang, Lu Qi, Rongkun Zheng, Xi Chen, Yi Wang, Yu Qiao","submitted_at":"2024-07-18T17:59:17Z","abstract_excerpt":"Recent efforts in video reasoning segmentation (VRS) integrate large language models (LLMs) with perception models to localize and track objects via textual instructions, achieving barely satisfactory results in simple scenarios. However, they struggled to discriminate and deduce the objects from user queries in more real-world scenes featured by long durations, multiple objects, rapid motion, and heavy occlusions. In this work, we analyze the underlying causes of these limitations, and present ViLLa: Video reasoning segmentation with Large Language Model. Remarkably, our ViLLa manages to tack"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.14500","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.14500/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.14500","created_at":"2026-07-05T10:31:45.275609+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.14500v3","created_at":"2026-07-05T10:31:45.275609+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.14500","created_at":"2026-07-05T10:31:45.275609+00:00"},{"alias_kind":"pith_short_12","alias_value":"AUMPF7GYLRMF","created_at":"2026-07-05T10:31:45.275609+00:00"},{"alias_kind":"pith_short_16","alias_value":"AUMPF7GYLRMF4R5L","created_at":"2026-07-05T10:31:45.275609+00:00"},{"alias_kind":"pith_short_8","alias_value":"AUMPF7GY","created_at":"2026-07-05T10:31:45.275609+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":239,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07334","citing_title":"RCoT-Seg: Reinforced Chain-of-Thought for Video Reasoning and Segmentation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22836","citing_title":"AgentRVOS for MeViS-Text Track of 5th PVUW Challenge: 3rd Method","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15670","citing_title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17797","citing_title":"Weakly-Supervised Referring Video Object Segmentation through Text Supervision","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18665","citing_title":"APRVOS: 1st Place Winner of 5th PVUW MeViS-Audio Track","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS","json":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS.json","graph_json":"https://pith.science/api/pith-number/AUMPF7GYLRMF4R5LJKAGHYFXIS/graph.json","events_json":"https://pith.science/api/pith-number/AUMPF7GYLRMF4R5LJKAGHYFXIS/events.json","paper":"https://pith.science/paper/AUMPF7GY"},"agent_actions":{"view_html":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS","download_json":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS.json","view_paper":"https://pith.science/paper/AUMPF7GY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.14500&json=true","fetch_graph":"https://pith.science/api/pith-number/AUMPF7GYLRMF4R5LJKAGHYFXIS/graph.json","fetch_events":"https://pith.science/api/pith-number/AUMPF7GYLRMF4R5LJKAGHYFXIS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS/action/storage_attestation","attest_author":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS/action/author_attestation","sign_citation":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS/action/citation_signature","submit_replication":"https://pith.science/pith/AUMPF7GYLRMF4R5LJKAGHYFXIS/action/replication_record"}},"created_at":"2026-07-05T10:31:45.275609+00:00","updated_at":"2026-07-05T10:31:45.275609+00:00"}