{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YSHAW2SIILF2US2FOPRBDY7B5T","short_pith_number":"pith:YSHAW2SI","schema_version":"1.0","canonical_sha256":"c48e0b6a4842cbaa4b4573e211e3e1ecf55ae8156901c0c5dc384adeb174014b","source":{"kind":"arxiv","id":"2311.17072","version":2},"attestation_state":"computed","paper":{"title":"IG Captioner: Information Gain Captioners are Strong Zero-shot Classifiers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Chenglin Yang, Jiahui Yu, Siyuan Qiao, Tao Zhu, Yuan Cao, Yu Zhang","submitted_at":"2023-11-27T19:00:06Z","abstract_excerpt":"Generative training has been demonstrated to be powerful for building visual-language models. However, on zero-shot discriminative benchmarks, there is still a performance gap between models trained with generative and discriminative objectives. In this paper, we aim to narrow this gap by improving the efficacy of generative training on classification tasks, without any finetuning processes or additional modules.\n  Specifically, we focus on narrowing the gap between the generative captioner and the CLIP classifier. We begin by analysing the predictions made by the captioner and classifier and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17072","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-27T19:00:06Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM"],"title_canon_sha256":"954bdddd29bcd647e15531938f027b2860e2a4ec938c1f63598b24e44579e0ff","abstract_canon_sha256":"55d98bfbfd33406442ffd1a1316e75fa169ad99d769d53a5df782d855afab59b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:49.364599Z","signature_b64":"+TGNsixC0IstIDi7x70vumu4cnXivcmVzWwrlCjzsnhkjEg+NiW1ecZSjaB76XrCqdwn0q4bFquce+OUE8ImDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c48e0b6a4842cbaa4b4573e211e3e1ecf55ae8156901c0c5dc384adeb174014b","last_reissued_at":"2026-07-05T08:44:49.364101Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:49.364101Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"IG Captioner: Information Gain Captioners are Strong Zero-shot Classifiers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Chenglin Yang, Jiahui Yu, Siyuan Qiao, Tao Zhu, Yuan Cao, Yu Zhang","submitted_at":"2023-11-27T19:00:06Z","abstract_excerpt":"Generative training has been demonstrated to be powerful for building visual-language models. However, on zero-shot discriminative benchmarks, there is still a performance gap between models trained with generative and discriminative objectives. In this paper, we aim to narrow this gap by improving the efficacy of generative training on classification tasks, without any finetuning processes or additional modules.\n  Specifically, we focus on narrowing the gap between the generative captioner and the CLIP classifier. We begin by analysing the predictions made by the captioner and classifier and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17072","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17072","created_at":"2026-07-05T08:44:49.364158+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17072v2","created_at":"2026-07-05T08:44:49.364158+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17072","created_at":"2026-07-05T08:44:49.364158+00:00"},{"alias_kind":"pith_short_12","alias_value":"YSHAW2SIILF2","created_at":"2026-07-05T08:44:49.364158+00:00"},{"alias_kind":"pith_short_16","alias_value":"YSHAW2SIILF2US2F","created_at":"2026-07-05T08:44:49.364158+00:00"},{"alias_kind":"pith_short_8","alias_value":"YSHAW2SI","created_at":"2026-07-05T08:44:49.364158+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.10745","citing_title":"Bridging the Skeleton-Text Modality Gap: Diffusion-Powered Modality Alignment for Zero-shot Skeleton-based Action Recognition","ref_index":72,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T","json":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T.json","graph_json":"https://pith.science/api/pith-number/YSHAW2SIILF2US2FOPRBDY7B5T/graph.json","events_json":"https://pith.science/api/pith-number/YSHAW2SIILF2US2FOPRBDY7B5T/events.json","paper":"https://pith.science/paper/YSHAW2SI"},"agent_actions":{"view_html":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T","download_json":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T.json","view_paper":"https://pith.science/paper/YSHAW2SI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17072&json=true","fetch_graph":"https://pith.science/api/pith-number/YSHAW2SIILF2US2FOPRBDY7B5T/graph.json","fetch_events":"https://pith.science/api/pith-number/YSHAW2SIILF2US2FOPRBDY7B5T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T/action/storage_attestation","attest_author":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T/action/author_attestation","sign_citation":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T/action/citation_signature","submit_replication":"https://pith.science/pith/YSHAW2SIILF2US2FOPRBDY7B5T/action/replication_record"}},"created_at":"2026-07-05T08:44:49.364158+00:00","updated_at":"2026-07-05T08:44:49.364158+00:00"}