{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NHGDFX3S2F37VN4P7AGEXENIBJ","short_pith_number":"pith:NHGDFX3S","schema_version":"1.0","canonical_sha256":"69cc32df72d177fab78ff80c4b91a80a5d698483b6ed8892395762900a7495e4","source":{"kind":"arxiv","id":"2312.06685","version":1},"attestation_state":"computed","paper":{"title":"Causal-CoG: A Causal-Effect Look at Context Generation for Boosting Multi-modal Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alan Yuille, Shitian Zhao, Yadong Lu, Yan Wang, Zhuowan Li","submitted_at":"2023-12-09T08:44:41Z","abstract_excerpt":"While Multi-modal Language Models (MLMs) demonstrate impressive multimodal ability, they still struggle on providing factual and precise responses for tasks like visual question answering (VQA). In this paper, we address this challenge from the perspective of contextual information. We propose Causal Context Generation, Causal-CoG, which is a prompting strategy that engages contextual information to enhance precise VQA during inference. Specifically, we prompt MLMs to generate contexts, i.e, text description of an image, and engage the generated contexts for question answering. Moreover, we in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.06685","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-09T08:44:41Z","cross_cats_sorted":[],"title_canon_sha256":"de38caf59bedd84c489a335ea0f466ab2b620180fc8f72389dc36fef6d3d9d8d","abstract_canon_sha256":"960436e1b91e62d8759b0e811a5356f2a94aa06e468cf77d5ae0ae7fe259369f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:22:51.519288Z","signature_b64":"uT10dQR/LDVf6jRCMPpen8MdBBuZz+OB7SoKIMdNTeVeuL2rEbVvCXfkaikTs4Yct/+8EE+Dauk1qyXZ/ZFSBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69cc32df72d177fab78ff80c4b91a80a5d698483b6ed8892395762900a7495e4","last_reissued_at":"2026-07-05T07:22:51.518823Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:22:51.518823Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal-CoG: A Causal-Effect Look at Context Generation for Boosting Multi-modal Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alan Yuille, Shitian Zhao, Yadong Lu, Yan Wang, Zhuowan Li","submitted_at":"2023-12-09T08:44:41Z","abstract_excerpt":"While Multi-modal Language Models (MLMs) demonstrate impressive multimodal ability, they still struggle on providing factual and precise responses for tasks like visual question answering (VQA). In this paper, we address this challenge from the perspective of contextual information. We propose Causal Context Generation, Causal-CoG, which is a prompting strategy that engages contextual information to enhance precise VQA during inference. Specifically, we prompt MLMs to generate contexts, i.e, text description of an image, and engage the generated contexts for question answering. Moreover, we in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.06685","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.06685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.06685","created_at":"2026-07-05T07:22:51.518883+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.06685v1","created_at":"2026-07-05T07:22:51.518883+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.06685","created_at":"2026-07-05T07:22:51.518883+00:00"},{"alias_kind":"pith_short_12","alias_value":"NHGDFX3S2F37","created_at":"2026-07-05T07:22:51.518883+00:00"},{"alias_kind":"pith_short_16","alias_value":"NHGDFX3S2F37VN4P","created_at":"2026-07-05T07:22:51.518883+00:00"},{"alias_kind":"pith_short_8","alias_value":"NHGDFX3S","created_at":"2026-07-05T07:22:51.518883+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.11600","citing_title":"GraphRAG-Causal: A novel graph-augmented framework for causal reasoning and annotation in news","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ","json":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ.json","graph_json":"https://pith.science/api/pith-number/NHGDFX3S2F37VN4P7AGEXENIBJ/graph.json","events_json":"https://pith.science/api/pith-number/NHGDFX3S2F37VN4P7AGEXENIBJ/events.json","paper":"https://pith.science/paper/NHGDFX3S"},"agent_actions":{"view_html":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ","download_json":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ.json","view_paper":"https://pith.science/paper/NHGDFX3S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.06685&json=true","fetch_graph":"https://pith.science/api/pith-number/NHGDFX3S2F37VN4P7AGEXENIBJ/graph.json","fetch_events":"https://pith.science/api/pith-number/NHGDFX3S2F37VN4P7AGEXENIBJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ/action/storage_attestation","attest_author":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ/action/author_attestation","sign_citation":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ/action/citation_signature","submit_replication":"https://pith.science/pith/NHGDFX3S2F37VN4P7AGEXENIBJ/action/replication_record"}},"created_at":"2026-07-05T07:22:51.518883+00:00","updated_at":"2026-07-05T07:22:51.518883+00:00"}