{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:X5KFCWHNLIBCLXB3VYLCBDZDGC","short_pith_number":"pith:X5KFCWHN","schema_version":"1.0","canonical_sha256":"bf545158ed5a0225dc3bae16208f2330b8a558548c27643de34e5f9ec0abb716","source":{"kind":"arxiv","id":"2204.04908","version":2},"attestation_state":"computed","paper":{"title":"No Token Left Behind: Explainability-Aided Image Classification and Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hila Chefer, Lior Wolf, Roni Paiss","submitted_at":"2022-04-11T07:16:39Z","abstract_excerpt":"The application of zero-shot learning in computer vision has been revolutionized by the use of image-text matching models. The most notable example, CLIP, has been widely used for both zero-shot classification and guiding generative models with a text prompt. However, the zero-shot use of CLIP is unstable with respect to the phrasing of the input text, making it necessary to carefully engineer the prompts used. We find that this instability stems from a selective similarity score, which is based only on a subset of the semantically meaningful input tokens. To mitigate it, we present a novel ex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.04908","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-04-11T07:16:39Z","cross_cats_sorted":[],"title_canon_sha256":"0d913a40f71a24a891e3c77a9fbfcae51223cd5b8007aaff0731417475b0eaed","abstract_canon_sha256":"3166427ba3bce21b784bcaee0e7355baf29e7b1401bb28b3f67b6dd8118fd2c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:46:36.681128Z","signature_b64":"bP7DB6XBFy8k5oOjv5qpZ16BgWpH3gCDNXfJiM1voMVKZnj2NJnyxUsZsoy+/UV2v/3j2uG+MbxxZau6Ye5HBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf545158ed5a0225dc3bae16208f2330b8a558548c27643de34e5f9ec0abb716","last_reissued_at":"2026-07-05T04:46:36.680540Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:46:36.680540Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"No Token Left Behind: Explainability-Aided Image Classification and Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hila Chefer, Lior Wolf, Roni Paiss","submitted_at":"2022-04-11T07:16:39Z","abstract_excerpt":"The application of zero-shot learning in computer vision has been revolutionized by the use of image-text matching models. The most notable example, CLIP, has been widely used for both zero-shot classification and guiding generative models with a text prompt. However, the zero-shot use of CLIP is unstable with respect to the phrasing of the input text, making it necessary to carefully engineer the prompts used. We find that this instability stems from a selective similarity score, which is based only on a subset of the semantically meaningful input tokens. To mitigate it, we present a novel ex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.04908","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.04908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.04908","created_at":"2026-07-05T04:46:36.680608+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.04908v2","created_at":"2026-07-05T04:46:36.680608+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.04908","created_at":"2026-07-05T04:46:36.680608+00:00"},{"alias_kind":"pith_short_12","alias_value":"X5KFCWHNLIBC","created_at":"2026-07-05T04:46:36.680608+00:00"},{"alias_kind":"pith_short_16","alias_value":"X5KFCWHNLIBCLXB3","created_at":"2026-07-05T04:46:36.680608+00:00"},{"alias_kind":"pith_short_8","alias_value":"X5KFCWHN","created_at":"2026-07-05T04:46:36.680608+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2208.01618","citing_title":"An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC","json":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC.json","graph_json":"https://pith.science/api/pith-number/X5KFCWHNLIBCLXB3VYLCBDZDGC/graph.json","events_json":"https://pith.science/api/pith-number/X5KFCWHNLIBCLXB3VYLCBDZDGC/events.json","paper":"https://pith.science/paper/X5KFCWHN"},"agent_actions":{"view_html":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC","download_json":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC.json","view_paper":"https://pith.science/paper/X5KFCWHN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.04908&json=true","fetch_graph":"https://pith.science/api/pith-number/X5KFCWHNLIBCLXB3VYLCBDZDGC/graph.json","fetch_events":"https://pith.science/api/pith-number/X5KFCWHNLIBCLXB3VYLCBDZDGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC/action/storage_attestation","attest_author":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC/action/author_attestation","sign_citation":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC/action/citation_signature","submit_replication":"https://pith.science/pith/X5KFCWHNLIBCLXB3VYLCBDZDGC/action/replication_record"}},"created_at":"2026-07-05T04:46:36.680608+00:00","updated_at":"2026-07-05T04:46:36.680608+00:00"}