{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ARA5YNZ3AMM2RJ5BN2G5ZJR2EF","short_pith_number":"pith:ARA5YNZ3","schema_version":"1.0","canonical_sha256":"0441dc373b0319a8a7a16e8ddca63a2173c0662abc8a4248c6363f802026f01c","source":{"kind":"arxiv","id":"2312.08870","version":2},"attestation_state":"computed","paper":{"title":"Vista-LLaMA: Reducing Hallucination in Video Language Models via Equal Distance to Visual Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Ma, Heng Wang, Jiashi Feng, Xiaojie Jin, Yi Yang, Yuchen Xian","submitted_at":"2023-12-12T09:47:59Z","abstract_excerpt":"Recent advances in large video-language models have displayed promising outcomes in video comprehension. Current approaches straightforwardly convert video into language tokens and employ large language models for multi-modal tasks. However, this method often leads to the generation of irrelevant content, commonly known as \"hallucination\", as the length of the text increases and the impact of the video diminishes. To address this problem, we propose Vista-LLaMA, a novel framework that maintains the consistent distance between all visual tokens and any language tokens, irrespective of the gener"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.08870","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-12-12T09:47:59Z","cross_cats_sorted":[],"title_canon_sha256":"83b64f150f0143c0c382a45f2be8545aed6f21f3c0d07431cff6cf0edecf5572","abstract_canon_sha256":"b2d12e1b480bcaeb34766ae1a944a91479b39730dd165ee234200c0981e7bb37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:43.566920Z","signature_b64":"rvHWsXtGtdli5R704jwQck7dzOEii2lcRRxqeNSPq3RX+HC8wU61A7+Denh+nVqRk2/ncqJPqn5laYN1VXDJDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0441dc373b0319a8a7a16e8ddca63a2173c0662abc8a4248c6363f802026f01c","last_reissued_at":"2026-07-05T10:22:43.566232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:43.566232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vista-LLaMA: Reducing Hallucination in Video Language Models via Equal Distance to Visual Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Ma, Heng Wang, Jiashi Feng, Xiaojie Jin, Yi Yang, Yuchen Xian","submitted_at":"2023-12-12T09:47:59Z","abstract_excerpt":"Recent advances in large video-language models have displayed promising outcomes in video comprehension. Current approaches straightforwardly convert video into language tokens and employ large language models for multi-modal tasks. However, this method often leads to the generation of irrelevant content, commonly known as \"hallucination\", as the length of the text increases and the impact of the video diminishes. To address this problem, we propose Vista-LLaMA, a novel framework that maintains the consistent distance between all visual tokens and any language tokens, irrespective of the gener"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.08870","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.08870/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.08870","created_at":"2026-07-05T10:22:43.566306+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.08870v2","created_at":"2026-07-05T10:22:43.566306+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.08870","created_at":"2026-07-05T10:22:43.566306+00:00"},{"alias_kind":"pith_short_12","alias_value":"ARA5YNZ3AMM2","created_at":"2026-07-05T10:22:43.566306+00:00"},{"alias_kind":"pith_short_16","alias_value":"ARA5YNZ3AMM2RJ5B","created_at":"2026-07-05T10:22:43.566306+00:00"},{"alias_kind":"pith_short_8","alias_value":"ARA5YNZ3","created_at":"2026-07-05T10:22:43.566306+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.05067","citing_title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09158","citing_title":"FaVChat: Hierarchical Prompt-Query Guided Facial Video Understanding with Data-Efficient GRPO","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16994","citing_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07476","citing_title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF","json":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF.json","graph_json":"https://pith.science/api/pith-number/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/graph.json","events_json":"https://pith.science/api/pith-number/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/events.json","paper":"https://pith.science/paper/ARA5YNZ3"},"agent_actions":{"view_html":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF","download_json":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF.json","view_paper":"https://pith.science/paper/ARA5YNZ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.08870&json=true","fetch_graph":"https://pith.science/api/pith-number/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/graph.json","fetch_events":"https://pith.science/api/pith-number/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/action/storage_attestation","attest_author":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/action/author_attestation","sign_citation":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/action/citation_signature","submit_replication":"https://pith.science/pith/ARA5YNZ3AMM2RJ5BN2G5ZJR2EF/action/replication_record"}},"created_at":"2026-07-05T10:22:43.566306+00:00","updated_at":"2026-07-05T10:22:43.566306+00:00"}