{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:U3L7K7DK5UGO5O3UQEPHX3WZRS","short_pith_number":"pith:U3L7K7DK","schema_version":"1.0","canonical_sha256":"a6d7f57c6aed0ceebb74811e7beed98cbd13fb9106e6f8c9a27904767a22761b","source":{"kind":"arxiv","id":"2106.13043","version":1},"attestation_state":"computed","paper":{"title":"AudioCLIP: Extending CLIP to Image, Text and Audio","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CV","eess.AS"],"primary_cat":"cs.SD","authors_text":"Andreas Dengel, Andrey Guzhov, Federico Raue, J\\\"orn Hees","submitted_at":"2021-06-24T14:16:38Z","abstract_excerpt":"In the past, the rapidly evolving field of sound classification greatly benefited from the application of methods from other domains. Today, we observe the trend to fuse domain-specific tasks and approaches together, which provides the community with new outstanding models.\n  In this work, we present an extension of the CLIP model that handles audio in addition to text and images. Our proposed model incorporates the ESResNeXt audio-model into the CLIP framework using the AudioSet dataset. Such a combination enables the proposed model to perform bimodal and unimodal classification and querying,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.13043","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2021-06-24T14:16:38Z","cross_cats_sorted":["cs.CV","eess.AS"],"title_canon_sha256":"13615740ecd59b1e9663c9b6bcfe3e99d84b4e338269d3fe36771a48e82de21a","abstract_canon_sha256":"ad008d44a011bbd43492bc71781d975069c02c0a5bede636f0420f47262be43e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:55:48.825387Z","signature_b64":"H3M90MpiBsbfmz/COtPh8xaLFBHGUn5ruCnZZ+Z1+e9jBlG8s4qeusNaOyLTGjlAFRXbRuUvJwwZeyrqYjo3Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6d7f57c6aed0ceebb74811e7beed98cbd13fb9106e6f8c9a27904767a22761b","last_reissued_at":"2026-07-05T04:55:48.824931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:55:48.824931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AudioCLIP: Extending CLIP to Image, Text and Audio","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CV","eess.AS"],"primary_cat":"cs.SD","authors_text":"Andreas Dengel, Andrey Guzhov, Federico Raue, J\\\"orn Hees","submitted_at":"2021-06-24T14:16:38Z","abstract_excerpt":"In the past, the rapidly evolving field of sound classification greatly benefited from the application of methods from other domains. Today, we observe the trend to fuse domain-specific tasks and approaches together, which provides the community with new outstanding models.\n  In this work, we present an extension of the CLIP model that handles audio in addition to text and images. Our proposed model incorporates the ESResNeXt audio-model into the CLIP framework using the AudioSet dataset. Such a combination enables the proposed model to perform bimodal and unimodal classification and querying,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.13043","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.13043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.13043","created_at":"2026-07-05T04:55:48.824993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.13043v1","created_at":"2026-07-05T04:55:48.824993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.13043","created_at":"2026-07-05T04:55:48.824993+00:00"},{"alias_kind":"pith_short_12","alias_value":"U3L7K7DK5UGO","created_at":"2026-07-05T04:55:48.824993+00:00"},{"alias_kind":"pith_short_16","alias_value":"U3L7K7DK5UGO5O3U","created_at":"2026-07-05T04:55:48.824993+00:00"},{"alias_kind":"pith_short_8","alias_value":"U3L7K7DK","created_at":"2026-07-05T04:55:48.824993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.11043","citing_title":"EmergentBridge: Improving Zero-Shot Cross-Modal Transfer in Unified Multimodal Embedding Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11043","citing_title":"EmergentBridge: Improving Zero-Shot Cross-Modal Transfer in Unified Multimodal Embedding Models","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS","json":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS.json","graph_json":"https://pith.science/api/pith-number/U3L7K7DK5UGO5O3UQEPHX3WZRS/graph.json","events_json":"https://pith.science/api/pith-number/U3L7K7DK5UGO5O3UQEPHX3WZRS/events.json","paper":"https://pith.science/paper/U3L7K7DK"},"agent_actions":{"view_html":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS","download_json":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS.json","view_paper":"https://pith.science/paper/U3L7K7DK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.13043&json=true","fetch_graph":"https://pith.science/api/pith-number/U3L7K7DK5UGO5O3UQEPHX3WZRS/graph.json","fetch_events":"https://pith.science/api/pith-number/U3L7K7DK5UGO5O3UQEPHX3WZRS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS/action/storage_attestation","attest_author":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS/action/author_attestation","sign_citation":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS/action/citation_signature","submit_replication":"https://pith.science/pith/U3L7K7DK5UGO5O3UQEPHX3WZRS/action/replication_record"}},"created_at":"2026-07-05T04:55:48.824993+00:00","updated_at":"2026-07-05T04:55:48.824993+00:00"}