{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IDNPU42UXN5GIM6BNLV6JJFEOD","short_pith_number":"pith:IDNPU42U","schema_version":"1.0","canonical_sha256":"40dafa7354bb7a6433c16aebe4a4a470cc01f2b3f96cf119af6a065e9c826297","source":{"kind":"arxiv","id":"2312.01597","version":4},"attestation_state":"computed","paper":{"title":"SCLIP: Rethinking Self-Attention for Dense Vision-Language Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Feng Wang, Jieru Mei","submitted_at":"2023-12-04T03:18:46Z","abstract_excerpt":"Recent advances in contrastive language-image pretraining (CLIP) have demonstrated strong capabilities in zero-shot classification by aligning visual representations with target text embeddings in an image level. However, in dense prediction tasks, CLIP often struggles to localize visual features within an image and fails to give accurate pixel-level predictions, which prevents it from functioning as a generalized visual foundation model. In this work, we aim to enhance CLIP's potential for semantic segmentation with minimal modifications to its pretrained models. By rethinking self-attention,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.01597","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-04T03:18:46Z","cross_cats_sorted":[],"title_canon_sha256":"e248c0fc1a62d6cae8dce2b40a3aba79ca3706a35a9257862e07b6b9d057ae1c","abstract_canon_sha256":"aede19320a35eafaa7dc7d4bbc26161e8fb7c90dd49387d80bcdec61970302b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:26:23.057595Z","signature_b64":"aTqaFfUK1aIpClDApcgLjzry7kvLHv/GHFLOyTRS6NUfG1f8IyjOVG4BJumF1hk/Cc93/QIdNJKKRgp4QYOTAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40dafa7354bb7a6433c16aebe4a4a470cc01f2b3f96cf119af6a065e9c826297","last_reissued_at":"2026-07-05T09:26:23.057060Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:26:23.057060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SCLIP: Rethinking Self-Attention for Dense Vision-Language Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Feng Wang, Jieru Mei","submitted_at":"2023-12-04T03:18:46Z","abstract_excerpt":"Recent advances in contrastive language-image pretraining (CLIP) have demonstrated strong capabilities in zero-shot classification by aligning visual representations with target text embeddings in an image level. However, in dense prediction tasks, CLIP often struggles to localize visual features within an image and fails to give accurate pixel-level predictions, which prevents it from functioning as a generalized visual foundation model. In this work, we aim to enhance CLIP's potential for semantic segmentation with minimal modifications to its pretrained models. By rethinking self-attention,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.01597","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.01597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.01597","created_at":"2026-07-05T09:26:23.057136+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.01597v4","created_at":"2026-07-05T09:26:23.057136+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.01597","created_at":"2026-07-05T09:26:23.057136+00:00"},{"alias_kind":"pith_short_12","alias_value":"IDNPU42UXN5G","created_at":"2026-07-05T09:26:23.057136+00:00"},{"alias_kind":"pith_short_16","alias_value":"IDNPU42UXN5GIM6B","created_at":"2026-07-05T09:26:23.057136+00:00"},{"alias_kind":"pith_short_8","alias_value":"IDNPU42U","created_at":"2026-07-05T09:26:23.057136+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07135","citing_title":"Sparse Attention for Dense Open-Vocabulary Prediction in CLIP","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":241,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18885","citing_title":"LARE: Low-Attention Region Encoding for Text-Image Retrieval","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12632","citing_title":"TeD-Loc: Text Distillation for Weakly Supervised Object Localization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18193","citing_title":"Best Segmentation Buddies for Image-Shape Correspondence","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19410","citing_title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08730","citing_title":"SegEarth-OV3: Exploring SAM 3 for Open-Vocabulary Semantic Segmentation in Remote Sensing Images","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD","json":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD.json","graph_json":"https://pith.science/api/pith-number/IDNPU42UXN5GIM6BNLV6JJFEOD/graph.json","events_json":"https://pith.science/api/pith-number/IDNPU42UXN5GIM6BNLV6JJFEOD/events.json","paper":"https://pith.science/paper/IDNPU42U"},"agent_actions":{"view_html":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD","download_json":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD.json","view_paper":"https://pith.science/paper/IDNPU42U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.01597&json=true","fetch_graph":"https://pith.science/api/pith-number/IDNPU42UXN5GIM6BNLV6JJFEOD/graph.json","fetch_events":"https://pith.science/api/pith-number/IDNPU42UXN5GIM6BNLV6JJFEOD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD/action/storage_attestation","attest_author":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD/action/author_attestation","sign_citation":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD/action/citation_signature","submit_replication":"https://pith.science/pith/IDNPU42UXN5GIM6BNLV6JJFEOD/action/replication_record"}},"created_at":"2026-07-05T09:26:23.057136+00:00","updated_at":"2026-07-05T09:26:23.057136+00:00"}