{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:I3S5VDMQ6ZK7SYNDORFPFQUXEB","short_pith_number":"pith:I3S5VDMQ","schema_version":"1.0","canonical_sha256":"46e5da8d90f655f961a3744af2c297204ed7b7e73635310d594fc9c55c0c6470","source":{"kind":"arxiv","id":"2310.05872","version":2},"attestation_state":"computed","paper":{"title":"ViCor: Bridging Visual Understanding and Commonsense Reasoning with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Kaiwen Zhou, Kwonjoon Lee, Teruhisa Misu, Xin Eric Wang","submitted_at":"2023-10-09T17:10:35Z","abstract_excerpt":"In our work, we explore the synergistic capabilities of pre-trained vision-and-language models (VLMs) and large language models (LLMs) on visual commonsense reasoning (VCR) problems. We find that VLMs and LLMs-based decision pipelines are good at different kinds of VCR problems. Pre-trained VLMs exhibit strong performance for problems involving understanding the literal visual content, which we noted as visual commonsense understanding (VCU). For problems where the goal is to infer conclusions beyond image content, which we noted as visual commonsense inference (VCI), VLMs face difficulties, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.05872","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-09T17:10:35Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"fb20b8c0666b557e3e3580b49e2538cb0ab17d99cc2606c5acd785b1909422e5","abstract_canon_sha256":"dc7c6fef5036020e2804043c2435381c274284dee59589c21335b74095b4a36e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:08.097488Z","signature_b64":"xeyHMOHaSoqGAokGtc/iqshqTYlmDifgsz5Dqw518OE0EJzHMVnpTbMFa1q/22dxck3PVkRJ62uSuQIm+viMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"46e5da8d90f655f961a3744af2c297204ed7b7e73635310d594fc9c55c0c6470","last_reissued_at":"2026-07-05T08:20:08.097024Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:08.097024Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViCor: Bridging Visual Understanding and Commonsense Reasoning with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Kaiwen Zhou, Kwonjoon Lee, Teruhisa Misu, Xin Eric Wang","submitted_at":"2023-10-09T17:10:35Z","abstract_excerpt":"In our work, we explore the synergistic capabilities of pre-trained vision-and-language models (VLMs) and large language models (LLMs) on visual commonsense reasoning (VCR) problems. We find that VLMs and LLMs-based decision pipelines are good at different kinds of VCR problems. Pre-trained VLMs exhibit strong performance for problems involving understanding the literal visual content, which we noted as visual commonsense understanding (VCU). For problems where the goal is to infer conclusions beyond image content, which we noted as visual commonsense inference (VCI), VLMs face difficulties, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.05872","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.05872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.05872","created_at":"2026-07-05T08:20:08.097081+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.05872v2","created_at":"2026-07-05T08:20:08.097081+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.05872","created_at":"2026-07-05T08:20:08.097081+00:00"},{"alias_kind":"pith_short_12","alias_value":"I3S5VDMQ6ZK7","created_at":"2026-07-05T08:20:08.097081+00:00"},{"alias_kind":"pith_short_16","alias_value":"I3S5VDMQ6ZK7SYND","created_at":"2026-07-05T08:20:08.097081+00:00"},{"alias_kind":"pith_short_8","alias_value":"I3S5VDMQ","created_at":"2026-07-05T08:20:08.097081+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.08158","citing_title":"How Vision-Language Tasks Benefit from Large Pre-trained Models: A Survey","ref_index":89,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB","json":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB.json","graph_json":"https://pith.science/api/pith-number/I3S5VDMQ6ZK7SYNDORFPFQUXEB/graph.json","events_json":"https://pith.science/api/pith-number/I3S5VDMQ6ZK7SYNDORFPFQUXEB/events.json","paper":"https://pith.science/paper/I3S5VDMQ"},"agent_actions":{"view_html":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB","download_json":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB.json","view_paper":"https://pith.science/paper/I3S5VDMQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.05872&json=true","fetch_graph":"https://pith.science/api/pith-number/I3S5VDMQ6ZK7SYNDORFPFQUXEB/graph.json","fetch_events":"https://pith.science/api/pith-number/I3S5VDMQ6ZK7SYNDORFPFQUXEB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB/action/storage_attestation","attest_author":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB/action/author_attestation","sign_citation":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB/action/citation_signature","submit_replication":"https://pith.science/pith/I3S5VDMQ6ZK7SYNDORFPFQUXEB/action/replication_record"}},"created_at":"2026-07-05T08:20:08.097081+00:00","updated_at":"2026-07-05T08:20:08.097081+00:00"}