{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EWZYCC7OYKM7QR6XO2XQ3ALDF3","short_pith_number":"pith:EWZYCC7O","schema_version":"1.0","canonical_sha256":"25b3810beec299f847d776af0d81632ef6b5e7221133f0c0a684022ffb2f2a52","source":{"kind":"arxiv","id":"2502.20578","version":2},"attestation_state":"computed","paper":{"title":"Interpreting CLIP with Hierarchical Sparse Autoencoders","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hubert Baniecki, Przemyslaw Biecek, Vladimir Zaigrajew","submitted_at":"2025-02-27T22:39:13Z","abstract_excerpt":"Sparse autoencoders (SAEs) are useful for detecting and steering interpretable features in neural networks, with particular potential for understanding complex multimodal representations. Given their ability to uncover interpretable features, SAEs are particularly valuable for analyzing large-scale vision-language models (e.g., CLIP and SigLIP), which are fundamental building blocks in modern systems yet remain challenging to interpret and control. However, current SAE methods are limited by optimizing both reconstruction quality and sparsity simultaneously, as they rely on either activation s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20578","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-27T22:39:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"68fd904348259420cc1dc307f9c44216a7c63acc2153b939fe1e8d014a49f409","abstract_canon_sha256":"bfe3bb354acd14585919350fdee0b43cdd68f9b69520007ccbf8daf4c01fd806"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:53.495946Z","signature_b64":"d6LYql0+cYbwNiw4mMUI8Fg6bVkhJnNvjfN3OYfWQ87OPRctnYviwLp7TjvhbmSSszedlOplEBtyyeS1w5ZVCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25b3810beec299f847d776af0d81632ef6b5e7221133f0c0a684022ffb2f2a52","last_reissued_at":"2026-07-05T11:10:53.495401Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:53.495401Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpreting CLIP with Hierarchical Sparse Autoencoders","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hubert Baniecki, Przemyslaw Biecek, Vladimir Zaigrajew","submitted_at":"2025-02-27T22:39:13Z","abstract_excerpt":"Sparse autoencoders (SAEs) are useful for detecting and steering interpretable features in neural networks, with particular potential for understanding complex multimodal representations. Given their ability to uncover interpretable features, SAEs are particularly valuable for analyzing large-scale vision-language models (e.g., CLIP and SigLIP), which are fundamental building blocks in modern systems yet remain challenging to interpret and control. However, current SAE methods are limited by optimizing both reconstruction quality and sparsity simultaneously, as they rely on either activation s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20578","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20578/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20578","created_at":"2026-07-05T11:10:53.495469+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20578v2","created_at":"2026-07-05T11:10:53.495469+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20578","created_at":"2026-07-05T11:10:53.495469+00:00"},{"alias_kind":"pith_short_12","alias_value":"EWZYCC7OYKM7","created_at":"2026-07-05T11:10:53.495469+00:00"},{"alias_kind":"pith_short_16","alias_value":"EWZYCC7OYKM7QR6X","created_at":"2026-07-05T11:10:53.495469+00:00"},{"alias_kind":"pith_short_8","alias_value":"EWZYCC7O","created_at":"2026-07-05T11:10:53.495469+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22994","citing_title":"Do Sparse Autoencoders Learn Meaningful Concept Hierarchies?","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22902","citing_title":"Transcoders Trace Visual Grounding and Hallucinations in Vision-Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22679","citing_title":"Conceptualizing Embeddings: Sparse Disentanglement for Vision-Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11107","citing_title":"Birds of a Feather Flock Together: Background-Invariant Representations via Linear Structure in VLMs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00899","citing_title":"LatentDiff: Scaling Semantic Dataset Comparison to Millions of Images","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07802","citing_title":"Latent Anomaly Knowledge Excavation: Unveiling Sparse Sensitive Neurons in Vision-Language Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05724","citing_title":"Beyond Semantics: Disentangling Information Scope in Sparse Autoencoders for CLIP","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3","json":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3.json","graph_json":"https://pith.science/api/pith-number/EWZYCC7OYKM7QR6XO2XQ3ALDF3/graph.json","events_json":"https://pith.science/api/pith-number/EWZYCC7OYKM7QR6XO2XQ3ALDF3/events.json","paper":"https://pith.science/paper/EWZYCC7O"},"agent_actions":{"view_html":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3","download_json":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3.json","view_paper":"https://pith.science/paper/EWZYCC7O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20578&json=true","fetch_graph":"https://pith.science/api/pith-number/EWZYCC7OYKM7QR6XO2XQ3ALDF3/graph.json","fetch_events":"https://pith.science/api/pith-number/EWZYCC7OYKM7QR6XO2XQ3ALDF3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3/action/storage_attestation","attest_author":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3/action/author_attestation","sign_citation":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3/action/citation_signature","submit_replication":"https://pith.science/pith/EWZYCC7OYKM7QR6XO2XQ3ALDF3/action/replication_record"}},"created_at":"2026-07-05T11:10:53.495469+00:00","updated_at":"2026-07-05T11:10:53.495469+00:00"}