{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:QDAIDVHAC6EH4CMAYO3Q5FGSNT","short_pith_number":"pith:QDAIDVHA","schema_version":"1.0","canonical_sha256":"80c081d4e017887e0980c3b70e94d26cef776909e47ece5c9b2712d60afdcdab","source":{"kind":"arxiv","id":"2103.03493","version":1},"attestation_state":"computed","paper":{"title":"Causal Attention for Vision-Language Tasks","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guojun Qi, Hanwang Zhang, Jianfei Cai, Xu Yang","submitted_at":"2021-03-05T06:38:25Z","abstract_excerpt":"We present a novel attention mechanism: Causal Attention (CATT), to remove the ever-elusive confounding effect in existing attention-based vision-language models. This effect causes harmful bias that misleads the attention module to focus on the spurious correlations in training data, damaging the model generalization. As the confounder is unobserved in general, we use the front-door adjustment to realize the causal intervention, which does not require any knowledge on the confounder. Specifically, CATT is implemented as a combination of 1) In-Sample Attention (IS-ATT) and 2) Cross-Sample Atte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.03493","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2021-03-05T06:38:25Z","cross_cats_sorted":[],"title_canon_sha256":"0d0a38167b3cdaaf0274a6c1e9902bf3f9f14591eb48238e25702553ecbfa7cc","abstract_canon_sha256":"75c003663ab31991fccea843cca9036d677983e88b87fce33340ecdb7c8c4645"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:20:35.982824Z","signature_b64":"yl4nTkgYr/OYuswlyR4jK8rTDVY+LUjSRPs5Eaq90xBZy6WD+Hp3ZkYpIzW6hJaVvGnHNUiZJbIIOf8OSqhpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80c081d4e017887e0980c3b70e94d26cef776909e47ece5c9b2712d60afdcdab","last_reissued_at":"2026-07-05T02:20:35.982349Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:20:35.982349Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal Attention for Vision-Language Tasks","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guojun Qi, Hanwang Zhang, Jianfei Cai, Xu Yang","submitted_at":"2021-03-05T06:38:25Z","abstract_excerpt":"We present a novel attention mechanism: Causal Attention (CATT), to remove the ever-elusive confounding effect in existing attention-based vision-language models. This effect causes harmful bias that misleads the attention module to focus on the spurious correlations in training data, damaging the model generalization. As the confounder is unobserved in general, we use the front-door adjustment to realize the causal intervention, which does not require any knowledge on the confounder. Specifically, CATT is implemented as a combination of 1) In-Sample Attention (IS-ATT) and 2) Cross-Sample Atte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.03493","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.03493/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.03493","created_at":"2026-07-05T02:20:35.982408+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.03493v1","created_at":"2026-07-05T02:20:35.982408+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.03493","created_at":"2026-07-05T02:20:35.982408+00:00"},{"alias_kind":"pith_short_12","alias_value":"QDAIDVHAC6EH","created_at":"2026-07-05T02:20:35.982408+00:00"},{"alias_kind":"pith_short_16","alias_value":"QDAIDVHAC6EH4CMA","created_at":"2026-07-05T02:20:35.982408+00:00"},{"alias_kind":"pith_short_8","alias_value":"QDAIDVHA","created_at":"2026-07-05T02:20:35.982408+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20732","citing_title":"Deep Attention Reweighting: Post-Hoc Attention-Based Feature Aggregation in CNNs for Disentangling Core and Spurious Features under Spurious Correlations","ref_index":75,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT","json":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT.json","graph_json":"https://pith.science/api/pith-number/QDAIDVHAC6EH4CMAYO3Q5FGSNT/graph.json","events_json":"https://pith.science/api/pith-number/QDAIDVHAC6EH4CMAYO3Q5FGSNT/events.json","paper":"https://pith.science/paper/QDAIDVHA"},"agent_actions":{"view_html":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT","download_json":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT.json","view_paper":"https://pith.science/paper/QDAIDVHA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.03493&json=true","fetch_graph":"https://pith.science/api/pith-number/QDAIDVHAC6EH4CMAYO3Q5FGSNT/graph.json","fetch_events":"https://pith.science/api/pith-number/QDAIDVHAC6EH4CMAYO3Q5FGSNT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT/action/storage_attestation","attest_author":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT/action/author_attestation","sign_citation":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT/action/citation_signature","submit_replication":"https://pith.science/pith/QDAIDVHAC6EH4CMAYO3Q5FGSNT/action/replication_record"}},"created_at":"2026-07-05T02:20:35.982408+00:00","updated_at":"2026-07-05T02:20:35.982408+00:00"}