{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HQHSO3FCU2ITB3CLGZLD4TVZ5V","short_pith_number":"pith:HQHSO3FC","schema_version":"1.0","canonical_sha256":"3c0f276ca2a69130ec4b36563e4eb9ed47bc1207d0fa19857149940bf6ccc50a","source":{"kind":"arxiv","id":"2403.12801","version":1},"attestation_state":"computed","paper":{"title":"RelationVLM: Making Large Vision-Language Models Understand Visual Relations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baining Guo, Yan Lu, Zheng-Jun Zha, Zhipeng Huang, Zhizheng Zhang","submitted_at":"2024-03-19T15:01:19Z","abstract_excerpt":"The development of Large Vision-Language Models (LVLMs) is striving to catch up with the success of Large Language Models (LLMs), yet it faces more challenges to be resolved. Very recent works enable LVLMs to localize object-level visual contents and ground text to them. Nonetheless, current LVLMs still struggle to precisely understand visual relations due to the lack of relevant data. In this work, we present RelationVLM, a large vision-language model capable of comprehending various levels and types of relations whether across multiple images or within a video. Specifically, we devise a mult"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.12801","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-19T15:01:19Z","cross_cats_sorted":[],"title_canon_sha256":"57cbd69385164ac83b108ec9b7dd78418e3783d1866ce0c85ef8a56cfe93b4eb","abstract_canon_sha256":"a110e6b2f57288d56ee1f075c4a90639b11020c8fdbfd9f155ac0e0b21ee82c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:58:07.809495Z","signature_b64":"VGo7I0Y0TjrlmkTQEN7FAXG8w72R8kHoD8Pb1dywP1/Q+Q8XlAEJcCPga2DIrqmoXh7OzDz8aTMWyjQRNMHCBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c0f276ca2a69130ec4b36563e4eb9ed47bc1207d0fa19857149940bf6ccc50a","last_reissued_at":"2026-07-05T07:58:07.808973Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:58:07.808973Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RelationVLM: Making Large Vision-Language Models Understand Visual Relations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baining Guo, Yan Lu, Zheng-Jun Zha, Zhipeng Huang, Zhizheng Zhang","submitted_at":"2024-03-19T15:01:19Z","abstract_excerpt":"The development of Large Vision-Language Models (LVLMs) is striving to catch up with the success of Large Language Models (LLMs), yet it faces more challenges to be resolved. Very recent works enable LVLMs to localize object-level visual contents and ground text to them. Nonetheless, current LVLMs still struggle to precisely understand visual relations due to the lack of relevant data. In this work, we present RelationVLM, a large vision-language model capable of comprehending various levels and types of relations whether across multiple images or within a video. Specifically, we devise a mult"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.12801","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.12801/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.12801","created_at":"2026-07-05T07:58:07.809034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.12801v1","created_at":"2026-07-05T07:58:07.809034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.12801","created_at":"2026-07-05T07:58:07.809034+00:00"},{"alias_kind":"pith_short_12","alias_value":"HQHSO3FCU2IT","created_at":"2026-07-05T07:58:07.809034+00:00"},{"alias_kind":"pith_short_16","alias_value":"HQHSO3FCU2ITB3CL","created_at":"2026-07-05T07:58:07.809034+00:00"},{"alias_kind":"pith_short_8","alias_value":"HQHSO3FC","created_at":"2026-07-05T07:58:07.809034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21762","citing_title":"ViStruct: Simulating Expert-Like Reasoning Through Task Decomposition and Visual Attention Cues","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V","json":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V.json","graph_json":"https://pith.science/api/pith-number/HQHSO3FCU2ITB3CLGZLD4TVZ5V/graph.json","events_json":"https://pith.science/api/pith-number/HQHSO3FCU2ITB3CLGZLD4TVZ5V/events.json","paper":"https://pith.science/paper/HQHSO3FC"},"agent_actions":{"view_html":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V","download_json":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V.json","view_paper":"https://pith.science/paper/HQHSO3FC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.12801&json=true","fetch_graph":"https://pith.science/api/pith-number/HQHSO3FCU2ITB3CLGZLD4TVZ5V/graph.json","fetch_events":"https://pith.science/api/pith-number/HQHSO3FCU2ITB3CLGZLD4TVZ5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V/action/storage_attestation","attest_author":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V/action/author_attestation","sign_citation":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V/action/citation_signature","submit_replication":"https://pith.science/pith/HQHSO3FCU2ITB3CLGZLD4TVZ5V/action/replication_record"}},"created_at":"2026-07-05T07:58:07.809034+00:00","updated_at":"2026-07-05T07:58:07.809034+00:00"}