{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QJR3TPG7UYQUPKCXMZL56NQQJG","short_pith_number":"pith:QJR3TPG7","schema_version":"1.0","canonical_sha256":"8263b9bcdfa62147a8576657df3610498b99ce3aec500f10e0a247d77b7c5d30","source":{"kind":"arxiv","id":"2311.06612","version":1},"attestation_state":"computed","paper":{"title":"PerceptionGPT: Effectively Fusing Visual Perception into LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiahui Gao, Jipeng Zhang, Lewei Yao, Renjie Pi, Tong Zhang","submitted_at":"2023-11-11T16:59:20Z","abstract_excerpt":"The integration of visual inputs with large language models (LLMs) has led to remarkable advancements in multi-modal capabilities, giving rise to visual large language models (VLLMs). However, effectively harnessing VLLMs for intricate visual perception tasks remains a challenge. In this paper, we present a novel end-to-end framework named PerceptionGPT, which efficiently and effectively equips the VLLMs with visual perception abilities by leveraging the representation power of LLMs' token embedding. Our proposed method treats the token embedding of the LLM as the carrier of spatial informatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.06612","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-11T16:59:20Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ae27efc46f075999b9ab6f0b034dcac10cb40052250892e91d6dc6d2e9b10ff3","abstract_canon_sha256":"d7d5e73cb1a86903ddc3acea6a2bf69d5ba1297e916e8bf55af67dbd3c52cbc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:42.816481Z","signature_b64":"RQ8/GwiPLzUWP8o4QjGGgDaek21MKCK6wYTDe9HMToWY4TZBwM8rUaybX4WHBRS6kNhfQbJz0H0pvsQdlIOxDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8263b9bcdfa62147a8576657df3610498b99ce3aec500f10e0a247d77b7c5d30","last_reissued_at":"2026-07-05T07:11:42.815945Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:42.815945Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PerceptionGPT: Effectively Fusing Visual Perception into LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiahui Gao, Jipeng Zhang, Lewei Yao, Renjie Pi, Tong Zhang","submitted_at":"2023-11-11T16:59:20Z","abstract_excerpt":"The integration of visual inputs with large language models (LLMs) has led to remarkable advancements in multi-modal capabilities, giving rise to visual large language models (VLLMs). However, effectively harnessing VLLMs for intricate visual perception tasks remains a challenge. In this paper, we present a novel end-to-end framework named PerceptionGPT, which efficiently and effectively equips the VLLMs with visual perception abilities by leveraging the representation power of LLMs' token embedding. Our proposed method treats the token embedding of the LLM as the carrier of spatial informatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.06612","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.06612/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.06612","created_at":"2026-07-05T07:11:42.816006+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.06612v1","created_at":"2026-07-05T07:11:42.816006+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.06612","created_at":"2026-07-05T07:11:42.816006+00:00"},{"alias_kind":"pith_short_12","alias_value":"QJR3TPG7UYQU","created_at":"2026-07-05T07:11:42.816006+00:00"},{"alias_kind":"pith_short_16","alias_value":"QJR3TPG7UYQUPKCX","created_at":"2026-07-05T07:11:42.816006+00:00"},{"alias_kind":"pith_short_8","alias_value":"QJR3TPG7","created_at":"2026-07-05T07:11:42.816006+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.13484","citing_title":"MINGLE: VLMs for Semantically Complex Region Detection in Urban Scenes","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG","json":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG.json","graph_json":"https://pith.science/api/pith-number/QJR3TPG7UYQUPKCXMZL56NQQJG/graph.json","events_json":"https://pith.science/api/pith-number/QJR3TPG7UYQUPKCXMZL56NQQJG/events.json","paper":"https://pith.science/paper/QJR3TPG7"},"agent_actions":{"view_html":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG","download_json":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG.json","view_paper":"https://pith.science/paper/QJR3TPG7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.06612&json=true","fetch_graph":"https://pith.science/api/pith-number/QJR3TPG7UYQUPKCXMZL56NQQJG/graph.json","fetch_events":"https://pith.science/api/pith-number/QJR3TPG7UYQUPKCXMZL56NQQJG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG/action/storage_attestation","attest_author":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG/action/author_attestation","sign_citation":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG/action/citation_signature","submit_replication":"https://pith.science/pith/QJR3TPG7UYQUPKCXMZL56NQQJG/action/replication_record"}},"created_at":"2026-07-05T07:11:42.816006+00:00","updated_at":"2026-07-05T07:11:42.816006+00:00"}