{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PQVEUYBKZZLIKESPQEKJN4LPFN","short_pith_number":"pith:PQVEUYBK","schema_version":"1.0","canonical_sha256":"7c2a4a602ace5685124f811496f16f2b62e83ea91bb8bf8e7fcdd8b09b556f28","source":{"kind":"arxiv","id":"2412.19806","version":1},"attestation_state":"computed","paper":{"title":"Vitron: A Unified Pixel-level Vision LLM for Understanding, Generating, Segmenting, Editing","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Hao Fei, Shengqiong Wu, Shuicheng Yan, Tat-Seng Chua","submitted_at":"2024-10-08T08:39:04Z","abstract_excerpt":"Recent developments of vision large language models (LLMs) have seen remarkable progress, yet still encounter challenges towards multimodal generalists, such as coarse-grained instance-level understanding, lack of unified support for both images and videos, and insufficient coverage across various vision tasks. In this paper, we present VITRON, a universal pixel-level vision LLM designed for comprehensive understanding, generating, segmenting, and editing of both static images and dynamic videos. Building on top of an LLM backbone, VITRON incorporates encoders for images, videos, and pixel-lev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.19806","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-08T08:39:04Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"edb013e4fd45b3522538383387f771f62a40071a7f3784629ba8d4de5c809a78","abstract_canon_sha256":"7552bae0d1248b8b86daa2e8f6cbc1fa600c9e7279ccd5d77d55a2a2f935cd1b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:49.719489Z","signature_b64":"0BMg0hDwWs/eV2U22u1PyGnhYJJbYCcMRrGvr27SFhZ00zMyhBWLgKYuylT1cpEn1icuLI/yZqXX6VntLiIYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c2a4a602ace5685124f811496f16f2b62e83ea91bb8bf8e7fcdd8b09b556f28","last_reissued_at":"2026-07-05T09:54:49.719073Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:49.719073Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vitron: A Unified Pixel-level Vision LLM for Understanding, Generating, Segmenting, Editing","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Hao Fei, Shengqiong Wu, Shuicheng Yan, Tat-Seng Chua","submitted_at":"2024-10-08T08:39:04Z","abstract_excerpt":"Recent developments of vision large language models (LLMs) have seen remarkable progress, yet still encounter challenges towards multimodal generalists, such as coarse-grained instance-level understanding, lack of unified support for both images and videos, and insufficient coverage across various vision tasks. In this paper, we present VITRON, a universal pixel-level vision LLM designed for comprehensive understanding, generating, segmenting, and editing of both static images and dynamic videos. Building on top of an LLM backbone, VITRON incorporates encoders for images, videos, and pixel-lev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.19806","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.19806/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.19806","created_at":"2026-07-05T09:54:49.719132+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.19806v1","created_at":"2026-07-05T09:54:49.719132+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.19806","created_at":"2026-07-05T09:54:49.719132+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQVEUYBKZZLI","created_at":"2026-07-05T09:54:49.719132+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQVEUYBKZZLIKESP","created_at":"2026-07-05T09:54:49.719132+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQVEUYBK","created_at":"2026-07-05T09:54:49.719132+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2511.10287","citing_title":"OutSafe-Bench: A Benchmark for Multimodal Offensive Content Detection in Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20275","citing_title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN","json":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN.json","graph_json":"https://pith.science/api/pith-number/PQVEUYBKZZLIKESPQEKJN4LPFN/graph.json","events_json":"https://pith.science/api/pith-number/PQVEUYBKZZLIKESPQEKJN4LPFN/events.json","paper":"https://pith.science/paper/PQVEUYBK"},"agent_actions":{"view_html":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN","download_json":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN.json","view_paper":"https://pith.science/paper/PQVEUYBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.19806&json=true","fetch_graph":"https://pith.science/api/pith-number/PQVEUYBKZZLIKESPQEKJN4LPFN/graph.json","fetch_events":"https://pith.science/api/pith-number/PQVEUYBKZZLIKESPQEKJN4LPFN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN/action/storage_attestation","attest_author":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN/action/author_attestation","sign_citation":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN/action/citation_signature","submit_replication":"https://pith.science/pith/PQVEUYBKZZLIKESPQEKJN4LPFN/action/replication_record"}},"created_at":"2026-07-05T09:54:49.719132+00:00","updated_at":"2026-07-05T09:54:49.719132+00:00"}